Compare commits
18 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 392eab4127 | |||
| 30b3f0ff75 | |||
| 01922291f6 | |||
| eb63e5e138 | |||
| 9029d48584 | |||
| 44b2eb2cc3 | |||
| 1cb09525e3 | |||
| c4833067f0 | |||
| 2bc8f336a0 | |||
| a955356821 | |||
| db474ca909 | |||
| a23e88c16d | |||
| def0f68d00 | |||
| fb86a18988 | |||
| b848143471 | |||
| 942517c7a8 | |||
| 7665e5e52f | |||
| bcf1a3e99b |
@@ -35,7 +35,7 @@ jobs:
|
|||||||
# Runs inside the toolchain image rather than installing Go per job. Note this puts
|
# Runs inside the toolchain image rather than installing Go per job. Note this puts
|
||||||
# the job on the dind bridge, which cannot reach github.com or get.helm.sh --
|
# the job on the dind bridge, which cannot reach github.com or get.helm.sh --
|
||||||
# proxy.golang.org and git.ryuvia.com are reachable, which is all this job needs.
|
# proxy.golang.org and git.ryuvia.com are reachable, which is all this job needs.
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
# act_runner destroys a job's own volumes when it finishes, so without these every
|
# act_runner destroys a job's own volumes when it finishes, so without these every
|
||||||
# run re-downloads the whole module graph. The names must appear in the runner's
|
# run re-downloads the whole module graph. The names must appear in the runner's
|
||||||
# container.valid_volumes allowlist (charts/act-runner in the k8s repo); unlisted
|
# container.valid_volumes allowlist (charts/act-runner in the k8s repo); unlisted
|
||||||
@@ -46,8 +46,8 @@ jobs:
|
|||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
- gobin-cache:/go/bin
|
- gobin-cache:/go/bin
|
||||||
|
|
||||||
# The suite needs a real Postgres -- there is no in-memory Postgres the way there was
|
# The suite needs a real Postgres -- there is no in-memory Postgres,
|
||||||
# an in-memory SQLite, so each test gets its own schema on a shared server instead.
|
# so each test gets its own schema on a shared server instead.
|
||||||
# The job and the service share the dind bridge, so the service is reachable by its
|
# The job and the service share the dind bridge, so the service is reachable by its
|
||||||
# name rather than on localhost.
|
# name rather than on localhost.
|
||||||
services:
|
services:
|
||||||
@@ -101,7 +101,7 @@ jobs:
|
|||||||
security:
|
security:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
container:
|
container:
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
volumes:
|
volumes:
|
||||||
- go-mod-cache:/go/pkg/mod
|
- go-mod-cache:/go/pkg/mod
|
||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ jobs:
|
|||||||
test:
|
test:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
container:
|
container:
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
volumes:
|
volumes:
|
||||||
- go-mod-cache:/go/pkg/mod
|
- go-mod-cache:/go/pkg/mod
|
||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
@@ -71,7 +71,7 @@ jobs:
|
|||||||
needs: test
|
needs: test
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
container:
|
container:
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
volumes:
|
volumes:
|
||||||
- go-mod-cache:/go/pkg/mod
|
- go-mod-cache:/go/pkg/mod
|
||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
|
|||||||
@@ -7,8 +7,6 @@
|
|||||||
# one (which has an unreachable entry) cannot abort a release
|
# one (which has an unreachable entry) cannot abort a release
|
||||||
/.helm-repos.yaml
|
/.helm-repos.yaml
|
||||||
|
|
||||||
# SQLite database files
|
|
||||||
*.db
|
|
||||||
*.db-shm
|
*.db-shm
|
||||||
*.db-wal
|
*.db-wal
|
||||||
|
|
||||||
|
|||||||
@@ -12,9 +12,9 @@ Preconditions and the plan, without side effects:
|
|||||||
```
|
```
|
||||||
|
|
||||||
Config is `.release.conf` here plus `make release-vars`. The process itself lives in
|
Config is `.release.conf` here plus `make release-vars`. The process itself lives in
|
||||||
`~/.claude/skills/release/`; why it is shaped this way is in README.md §Releasing.
|
`~/.claude/skills/release/`; why it is shaped this way is in docs/development.md (Releasing).
|
||||||
|
|
||||||
Two things about this repo specifically:
|
Three things about this repo specifically:
|
||||||
|
|
||||||
- **The image is scanned after it is published, not before.** `scan-image` runs trivy
|
- **The image is scanned after it is published, not before.** `scan-image` runs trivy
|
||||||
against the pushed image, because trivy cannot read a locally built one on this runner.
|
against the pushed image, because trivy cannot read a locally built one on this runner.
|
||||||
@@ -25,6 +25,16 @@ Two things about this repo specifically:
|
|||||||
so `chart-bump` needs `--image "$IMAGE"` to know which one moves. That sidecar backs up
|
so `chart-bump` needs `--image "$IMAGE"` to know which one moves. That sidecar backs up
|
||||||
SQLite; the Postgres move (#2) retires it in favour of a `postgresql` CR with a k8up
|
SQLite; the Postgres move (#2) retires it in favour of a `postgresql` CR with a k8up
|
||||||
`pg_dump` annotation, after which only the app image's tag is left.
|
`pg_dump` annotation, after which only the app image's tag is left.
|
||||||
|
- **Two demos pin this image, and `chart-bump` moves neither.** `terdut-demo` in
|
||||||
|
`Ryuvia/charts` is a `TerdutServer` CR that terdut-operator reconciles, and its
|
||||||
|
`values.yaml` `image.tag` is meant to match production's pin (same digest). The kind demo
|
||||||
|
in terdut-operator (`examples/demo/01-server.yaml`) pins a tag too. A release only bumps
|
||||||
|
the `terdut-server` wrapper, so both drift silently: `terdut-demo` sat at v0.37.0 through
|
||||||
|
v0.41.0-v0.43.0 until it was synced on 2026-10-08. After a release, bump `terdut-demo`'s
|
||||||
|
tag to the same `image-digest` and its `Chart.yaml` `version:` (Flux reconciles on
|
||||||
|
ChartVersion), as its own PR, and say in the release report whether you did. Neither
|
||||||
|
demo has anything but the pin to change, but read the version range's migrations first:
|
||||||
|
the demo's Postgres migrates forward at startup.
|
||||||
|
|
||||||
## Checks
|
## Checks
|
||||||
|
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ help: ## Show this help
|
|||||||
# -race below. Both need a Postgres to test against; see test-db.
|
# -race below. Both need a Postgres to test against; see test-db.
|
||||||
|
|
||||||
# The suite needs a Postgres, because the server does: there is no in-memory
|
# The suite needs a Postgres, because the server does: there is no in-memory
|
||||||
# Postgres the way there was an in-memory SQLite. TERDUT_TEST_DSN says where, and
|
# Postgres. TERDUT_TEST_DSN says where, and
|
||||||
# the tests fail rather than skip without it — a suite that quietly tests nothing
|
# the tests fail rather than skip without it — a suite that quietly tests nothing
|
||||||
# is worse than one that does not run. `make test-db` starts a local one;
|
# is worse than one that does not run. `make test-db` starts a local one;
|
||||||
# ci.yaml runs the same thing as a service container.
|
# ci.yaml runs the same thing as a service container.
|
||||||
|
|||||||
@@ -1,263 +1,60 @@
|
|||||||
# Service accounts: a scoped, non-human credential type
|
# Service accounts
|
||||||
|
|
||||||
This is a design note for a feature, not an implementation plan — it exists to
|
A non-human credential for automation (terdut-operator, CI, scripts). It is not a
|
||||||
propose the shape before writing code. It's raised directly by `terdut-operator`
|
`users` row: no password, no `is_admin`, no OIDC identity, so it can never be
|
||||||
(a separate repo, no shared code — see its `DESIGN.md` §6, §9, §13), which needs
|
pulled into login or group sync, and it is never mistaken for a person in an
|
||||||
a credential for unattended, repeatable API access and currently has no good one
|
audit trail. The bearer token has the same shape as an API key (SHA-256 hash
|
||||||
available. Anything automating terdut-server long-term (this operator, CI, future
|
stored, raw value shown once), prefixed `tdsa_`.
|
||||||
integrations) hits the same gap, so this is written as a general primitive, not
|
|
||||||
operator-specific.
|
|
||||||
|
|
||||||
## The problem
|
## Scopes
|
||||||
|
|
||||||
terdut-server has two credential types today, and neither fits "an unattended
|
- **instance** — acts as owner of every team's *configuration* (rename, OIDC
|
||||||
process that manages teams/schedules/policies on someone's behalf":
|
groups, escalation, dead man's switches, integrations, members, delete) and may
|
||||||
|
create teams. It is not a member of any team, so it reads no incidents or
|
||||||
|
queue. It is never an administrator: user management and
|
||||||
|
`/api/admin/settings` stay human-only.
|
||||||
|
- **team** — acts as owner of exactly one team, through a single synthetic
|
||||||
|
membership. It may also mint another service account for its own team.
|
||||||
|
|
||||||
- **User API keys** (`api_keys`, `internal/api/users.go`) are always tied to a
|
An account has many keys, so rotating is "mint a new key, revoke the old one"
|
||||||
real `users` row and carry that user's full rights — every team they're a
|
without losing the account's identity or history.
|
||||||
member of, their admin flag if set. There's no `kind`/`service` marker
|
|
||||||
distinguishing "a human's personal automation key" from "a login session," and
|
|
||||||
no way to mint one scoped to less than the full user.
|
|
||||||
- **Integration keys** (`integrations`, `internal/api/*teams*.go`) are team-scoped,
|
|
||||||
but narrowly: they authenticate exactly one inbound Alertmanager webhook call
|
|
||||||
(`POST /api/integrations/{key}/alertmanager`) and nothing else. They're not a
|
|
||||||
general management-API credential and shouldn't become one — overloading a
|
|
||||||
narrow, one-way ingestion credential with broad read/write access would weaken
|
|
||||||
the one property that makes it safe to embed in an Alertmanager config today.
|
|
||||||
|
|
||||||
The result: any automation that needs to create teams, set escalation policies,
|
## Endpoints
|
||||||
manage dead-man switches, or rotate integration keys has to hold a real human
|
|
||||||
admin's or team owner's API key. That key is exactly as powerful as that person
|
|
||||||
logging in — full team access, and full instance access if they're an admin.
|
|
||||||
`terdut-operator`'s design ran directly into this (its DESIGN.md §6): its
|
|
||||||
described bootstrap/rotation flow assumed a repeatable, identity-scoped way to
|
|
||||||
get a credential, and `/api/bootstrap`'s actual behavior (single-shot per
|
|
||||||
install, gated on `COUNT(*) FROM users`, confirmed via `internal/api/users.go`
|
|
||||||
and `charts/terdut-server/templates/bootstrap-job.yaml`) doesn't provide one —
|
|
||||||
it mints exactly one founding admin, once, ever.
|
|
||||||
|
|
||||||
## Goals
|
- `POST /api/service-accounts` `{name, scope, team_id}` — returns the account and
|
||||||
|
its first key. An instance-scoped account is granted by a human administrator;
|
||||||
|
a team-scoped one by an administrator, that team's owner, or an instance-scoped
|
||||||
|
account.
|
||||||
|
- `GET /api/service-accounts?name=` — look one up by name.
|
||||||
|
- `POST /api/service-accounts/{id}/keys`, `DELETE .../keys/{keyID}` — mint or
|
||||||
|
revoke a key. An instance-scoped account may manage any team-scoped account's
|
||||||
|
keys, and any account may manage its own.
|
||||||
|
|
||||||
- A credential type that isn't a human: doesn't touch OIDC group sync, login,
|
## Seeding the operator's account
|
||||||
session, or the `is_admin`/account-management semantics that come with a real
|
|
||||||
`users` row.
|
|
||||||
- Two scopes matching the two shapes automation actually needs: instance-wide
|
|
||||||
(create/list teams — what a server-owning controller needs) and team-scoped
|
|
||||||
(manage one team's escalation policy, dead-man switches, integrations,
|
|
||||||
schedule, OIDC group bindings — what a per-team controller or integration
|
|
||||||
needs).
|
|
||||||
- Repeatable issuance and rotation — unlike `/api/bootstrap`, callable more than
|
|
||||||
once, by anything that already holds admin rights, without destroying and
|
|
||||||
recreating state to get a fresh credential.
|
|
||||||
- Visibly distinct from a human in every place identity shows up (audit trails,
|
|
||||||
timeline entries, UI attribution) — a service account acting on a team should
|
|
||||||
never be indistinguishable from a person.
|
|
||||||
|
|
||||||
## Non-goals
|
`TERDUT_OPERATOR_KEY` (at least 32 characters) creates the instance-scoped account
|
||||||
|
`terdut-operator` if missing and replaces its `seed` key with this value at every
|
||||||
|
start (`internal/api/operator_key.go`). The deployer generates the key and
|
||||||
|
nothing has to call `/api/bootstrap` for it; rotating is a restart with a new
|
||||||
|
value. With `TERDUT_OPERATOR_MODE` on, configuration writes by humans are refused
|
||||||
|
and a service account of either scope passes.
|
||||||
|
|
||||||
- Not a general OAuth2/OIDC client-credentials flow — this is a bearer-token
|
## How it is enforced
|
||||||
primitive matching the shape `api_keys` already uses (SHA-256 hash stored,
|
|
||||||
raw key shown once at creation), not a new auth protocol.
|
|
||||||
- Not replacing integration keys — those stay as the narrow, one-way webhook
|
|
||||||
credential they are today.
|
|
||||||
- Not modeling per-endpoint or per-verb permissions within a scope — `instance`
|
|
||||||
and `team` are the only two scopes for now; finer-grained scoping is future
|
|
||||||
work if a real need shows up.
|
|
||||||
|
|
||||||
## Proposed shape
|
Every request resolves to one `Caller` (`internal/api/caller.go`): a human
|
||||||
|
(session or API key) or a service account.
|
||||||
|
|
||||||
### Schema
|
- `Caller.IsAdmin()` is true only for a human administrator. `AdminOnly` and
|
||||||
|
`requireSelfOrAdmin` key on it alone; do not widen them — each time a gap came up
|
||||||
|
the fix was a narrower purpose-built capability instead.
|
||||||
|
- `Caller.IsInstanceServiceAccount()` is true only for an instance-scoped account,
|
||||||
|
never for a human. `requireTeamOwner` and `callerOwnsTeam` admit it for any team.
|
||||||
|
- `Caller.Role(teamID)`/`TeamIDs()` are a human's memberships or a team-scoped
|
||||||
|
account's single owner membership; instance scope has none.
|
||||||
|
- `Caller.AsHuman()` is what a handler must call when it needs a real `user_id`;
|
||||||
|
handlers meant for people answer 403 to a service account instead of writing a
|
||||||
|
zero id.
|
||||||
|
|
||||||
```sql
|
Where a service account acts on an incident (acknowledge, resolve), the
|
||||||
CREATE TABLE service_accounts (
|
timeline and `acknowledged_by` record it through parallel `*_service_account_id`
|
||||||
id BIGSERIAL PRIMARY KEY,
|
columns, never as a user.
|
||||||
name TEXT NOT NULL UNIQUE, -- e.g. "terdut-operator"
|
|
||||||
scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
|
|
||||||
team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
-- team_id required iff scope = 'team'; NULL iff scope = 'instance'
|
|
||||||
created_by BIGINT REFERENCES users(id),
|
|
||||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE service_account_keys (
|
|
||||||
id BIGSERIAL PRIMARY KEY,
|
|
||||||
service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
|
|
||||||
key_hash TEXT NOT NULL UNIQUE,
|
|
||||||
name TEXT NOT NULL, -- e.g. "initial", "2026-Q4-rotation"
|
|
||||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
|
||||||
last_used_at TIMESTAMPTZ
|
|
||||||
);
|
|
||||||
```
|
|
||||||
|
|
||||||
Deliberately not a `users` row: no `password_hash`, no `is_admin`, no
|
|
||||||
`user_identities` linkage, so it's structurally impossible for a service account
|
|
||||||
to be pulled into OIDC group sync or password login. Multiple keys per account
|
|
||||||
(mirroring `api_keys`' existing one-user-many-keys shape) so rotation is "mint a
|
|
||||||
new key, revoke the old one," not "recreate the account."
|
|
||||||
|
|
||||||
### Endpoints
|
|
||||||
|
|
||||||
- `POST /api/service-accounts` — instance-scope/admin-only. Body:
|
|
||||||
`{"name": ..., "scope": "instance"|"team", "teamID": ... }` (teamID required
|
|
||||||
iff scope=team, and caller must be that team's owner or a system admin).
|
|
||||||
Returns the account plus its first raw key (shown once, same pattern as
|
|
||||||
`POST /api/users/{id}/api-keys`). Safe to call again with the same `name` —
|
|
||||||
see "idempotent lookup" below — unlike `/api/bootstrap`, which is inherently
|
|
||||||
one-shot by design (it's answering "does any user exist yet," a question with
|
|
||||||
no analogue once one already does).
|
|
||||||
- `POST /api/service-accounts/{id}/keys` — mint an additional key on an existing
|
|
||||||
account (self-service-equivalent: instance admin for `instance` scope, team
|
|
||||||
owner or system admin for `team` scope). Enables rotation without recreating
|
|
||||||
the account or losing its identity/audit history.
|
|
||||||
- `DELETE /api/service-accounts/{id}/keys/{keyID}` — revoke one key, mirroring
|
|
||||||
`DELETE /api/users/{id}/api-keys/{keyID}`.
|
|
||||||
- `GET /api/service-accounts?name=` — look up an existing account by name.
|
|
||||||
This is what turns "I tried to create my account and got a conflict" into a
|
|
||||||
normal flow instead of an error: a controller that expects to have already
|
|
||||||
registered itself calls this first, and only falls through to `POST` if
|
|
||||||
nothing comes back.
|
|
||||||
|
|
||||||
### Auth middleware
|
|
||||||
|
|
||||||
**Revised** (this section originally described an aspiration that didn't
|
|
||||||
match what shipped — `TEAM-LOOKUP.md` already caught one instance of that,
|
|
||||||
and a fuller audit found three more; this is the corrected, as-built
|
|
||||||
description, not the original proposal).
|
|
||||||
|
|
||||||
`internal/api/middleware.go`'s dual resolution (`Authorization: Bearer` →
|
|
||||||
`apiKeyUser()`, or session cookie → `sessionUser()`) and the service-account
|
|
||||||
path (`serviceAccountFor()`) both resolve into one `Caller` type
|
|
||||||
(`internal/api/caller.go`), not two parallel, un-unified context
|
|
||||||
representations the way an earlier version of this server kept them. Every
|
|
||||||
authorization predicate reads `Caller`'s methods:
|
|
||||||
|
|
||||||
- `Caller.IsAdmin()` — true **only** for a human system administrator, never
|
|
||||||
for a service account of either scope, under any circumstance. `AdminOnly`
|
|
||||||
and `requireSelfOrAdmin` key on this alone — user management
|
|
||||||
(`POST /api/users`, `PUT /api/users/{id}/admin`, etc.) and
|
|
||||||
`GET/PUT /api/admin/settings` stay human-only, forever. The original text
|
|
||||||
here claimed an instance-scoped service account satisfies `AdminOnly` "for
|
|
||||||
team-creation/listing purposes" — that was never true of the shipped code
|
|
||||||
(`TEAM-LOOKUP.md` caught the listing half; the creation half was always a
|
|
||||||
separate, bespoke check in `handleCreateTeam`, not `AdminOnly` itself) and
|
|
||||||
is not being made true now. Don't widen `AdminOnly`: every time this has
|
|
||||||
come up, the fix has been a narrower, purpose-built capability instead
|
|
||||||
(`?name=` lookups for teams and service accounts; now
|
|
||||||
`terdut-operator`'s own invite-minting feature for the one real gap this
|
|
||||||
boundary left — how a human ever gets a first login on a no-OIDC,
|
|
||||||
operator-managed install. See the bottom of "What this unblocks.")
|
|
||||||
- `Caller.IsInstanceServiceAccount()` — true only for an instance-scoped
|
|
||||||
service account, never for a human (including a human admin).
|
|
||||||
`handleCreateTeam` uses exactly this: a human creates a team by being a
|
|
||||||
human (and becomes its owner); an instance-scoped service account creates
|
|
||||||
one with no human owner at all. The two paths are not interchangeable, so
|
|
||||||
this predicate deliberately does not also admit a human admin.
|
|
||||||
- `Caller.Role(teamID)`/`TeamIDs()` — a human's real `team_members` rows, or
|
|
||||||
a team-scoped service account's single synthetic owner membership
|
|
||||||
(`serveAsServiceAccount`). This is what makes `requireTeamMember`/
|
|
||||||
`requireTeamOwner` treat a team-scoped service account as owner-equivalent
|
|
||||||
for that one team, with no separate branch needed in either function.
|
|
||||||
- `Caller.ServiceAccountID()` — used by `OperatorModeBlock` ("any service
|
|
||||||
account passes") and by `callerMayManageServiceAccount`'s self-rotation
|
|
||||||
check.
|
|
||||||
- `Caller.AsHuman()` — the accessor every handler that needs a real
|
|
||||||
`user_id` to act on behalf of must call and check, instead of reading a
|
|
||||||
user off context unconditionally. Before the `Caller` type existed, four
|
|
||||||
handlers did the latter and silently misbehaved for a service-account
|
|
||||||
caller: `handleMe` and `handleTestNotification` 500'd (a zero-value user id
|
|
||||||
that matches no row), `handleDismissOnboarding` silently no-op'd (`UPDATE
|
|
||||||
... WHERE id = 0` affects nothing, still returns 204), and
|
|
||||||
`handleCreateInvite` wrote that same zero value into `invites.created_by`
|
|
||||||
— a real foreign-key violation, not just a wrong answer, since that column
|
|
||||||
is nullable but was never passed as `nil`. All four now call `AsHuman()`
|
|
||||||
and return an explicit 403 ("this endpoint is for human accounts only")
|
|
||||||
or, for the invite case, leave `created_by` `NULL` the same way
|
|
||||||
`handleCreateServiceAccount` already did for the analogous situation.
|
|
||||||
|
|
||||||
**Team scope is owner-equivalent for every `requireTeamOwner` endpoint,
|
|
||||||
membership and invites included — by design, not by an unclosed gap.** An
|
|
||||||
earlier version of this document flagged this as "acknowledged rather than
|
|
||||||
closed," kept in check only by the social convention that nobody *builds*
|
|
||||||
automation against those two routes. That convention is retired:
|
|
||||||
`terdut-operator`'s `TerdutTeam` controller now mints and revokes its own
|
|
||||||
team's invite link through exactly this capability (its existing
|
|
||||||
team-scoped credential, `POST`/`DELETE /api/teams/{teamID}/invites`), which
|
|
||||||
is the real fix for the human-onboarding gap below — not a narrower
|
|
||||||
carve-out of this capability. `service_accounts_test.go`'s
|
|
||||||
`TestServiceAccount_TeamScopeManagesItsOwnInvites` pins it.
|
|
||||||
|
|
||||||
**A team-scoped account can also mint another service account scoped to its
|
|
||||||
own team** (`handleCreateServiceAccount`'s `callerOwnsTeam` branch, which a
|
|
||||||
team-scoped caller already satisfies for its own team via the synthetic
|
|
||||||
membership above). Kept, not restricted, for the same reason: a team-scoped
|
|
||||||
credential is that team's owner's reach, full stop — carving this one
|
|
||||||
capability out while leaving membership/invites alone would be an arbitrary
|
|
||||||
asymmetry. Pinned by
|
|
||||||
`TestServiceAccount_TeamScopeCanMintAnotherAccountForItsOwnTeam`.
|
|
||||||
|
|
||||||
**`callerMayManageServiceAccount` gained the one load-bearing fix this
|
|
||||||
redesign exists for:** an instance-scoped service account may manage
|
|
||||||
(mint/revoke a key on) *any* team-scoped account, not only one admin, that
|
|
||||||
team's human owner, or the account itself. `handleCreateServiceAccount`
|
|
||||||
already let an instance-scoped caller *create* a team-scoped account for
|
|
||||||
any team; this closes the gap where adopting or rotating one it didn't just
|
|
||||||
create in the same call — exactly `terdut-operator`'s documented
|
|
||||||
adopt-on-409 crash-window recovery (its own `DESIGN.md` §5) — 403'd forever
|
|
||||||
instead of succeeding (`terdut-operator#3`). Pinned by
|
|
||||||
`TestServiceAccount_InstanceScopeAdoptsAnExistingTeamScopedAccountsKey`.
|
|
||||||
|
|
||||||
Anywhere identity is recorded for a human (incident timeline
|
|
||||||
`acknowledged_by`/`assigned_to`, audit-relevant fields), a service-account
|
|
||||||
caller is still coerced into a bare `user_id` of `0` today — `Caller`'s new
|
|
||||||
`Identity()` accessor exists for exactly this follow-up, but wiring it in
|
|
||||||
needs a schema migration (an actor-attribution column distinct from
|
|
||||||
`user_id`) and is deliberately out of scope here. Tracked separately, not by
|
|
||||||
this document.
|
|
||||||
|
|
||||||
## What this unblocks
|
|
||||||
|
|
||||||
Directly resolves `terdut-operator` DESIGN.md §6's two broken assumptions:
|
|
||||||
1. **Bootstrap becomes single-purpose again.** `/api/bootstrap` mints exactly
|
|
||||||
the founding human admin, once. The operator's actual first-reconcile flow:
|
|
||||||
call `/api/bootstrap` only on a genuinely empty install; otherwise (or
|
|
||||||
immediately after, if it won the bootstrap race) call
|
|
||||||
`GET /api/service-accounts?name=terdut-operator`, and `POST` one if it
|
|
||||||
doesn't exist yet. From then on the operator never touches `/api/bootstrap`
|
|
||||||
again.
|
|
||||||
2. **Rotation becomes real.** `POST /api/service-accounts/{id}/keys` + revoke the
|
|
||||||
old one — no destructive DB-level workaround, no re-triggering a single-shot
|
|
||||||
endpoint that can't fire twice.
|
|
||||||
3. **Cross-namespace credential mirroring is no longer needed at all.**
|
|
||||||
`terdut-operator`'s current design holds every credential — instance- and
|
|
||||||
team-scoped alike — privately in the operator's own namespace, never in
|
|
||||||
the namespace of the CR each one authenticates for; reconciliation happens
|
|
||||||
entirely inside the operator's controller loop, so no CR owner ever needs
|
|
||||||
read access to a terdut-server credential regardless of same- or
|
|
||||||
cross-namespace `serverRef`. Team scoping is still what bounds the blast
|
|
||||||
radius of any individual credential: a leaked team-scoped key exposes
|
|
||||||
exactly one team's resources, never the whole server, which is what makes
|
|
||||||
holding many credentials in one place (the operator's namespace) an
|
|
||||||
acceptable trade rather than reintroducing the mirrored design's
|
|
||||||
server-admin-equivalent-everywhere problem.
|
|
||||||
4. **A human can get a first login on a no-OIDC, operator-managed install —
|
|
||||||
without ever touching `AdminOnly` or `/api/admin/settings`.** This was
|
|
||||||
filed as `terdut-server#23` ("no API path to create a human login after
|
|
||||||
bootstrap") and diagnosed, at the time, as this server needing to let a
|
|
||||||
service account through `AdminOnly`. It doesn't: the fix lives entirely
|
|
||||||
in `terdut-operator`, because a team-scoped credential was *already*
|
|
||||||
owner-equivalent for `POST /api/teams/{teamID}/invites`, and invite
|
|
||||||
redemption (`POST /api/signup` with an `invite` token) bypasses
|
|
||||||
`signup_mode` entirely — `terdut-operator` just never grew a feature to
|
|
||||||
use either fact. Its `TerdutTeam` controller now mints and surfaces one
|
|
||||||
via its own existing team-scoped credential (`spec.invite`,
|
|
||||||
`status.inviteSecretRef`, see that repo's own docs), so a human joins a
|
|
||||||
CRD-managed team by a real invite link, the same way anyone else would.
|
|
||||||
`terdut-server#23` is closed with this note once that feature ships — its
|
|
||||||
named routes stay human-only, correctly, not a gap.
|
|
||||||
|
|
||||||
## Suggested sequencing
|
|
||||||
|
|
||||||
Land this before `terdut-operator` implements any bootstrap/credential-handling
|
|
||||||
code — that code would otherwise be written against the current one-shot,
|
|
||||||
user-only credential model as a known-temporary workaround, which is wasted
|
|
||||||
effort on a repo that currently has zero implementation to begin with.
|
|
||||||
|
|||||||
@@ -1,73 +0,0 @@
|
|||||||
# Team lookup for service accounts: closing terdut-operator's create-path crash window
|
|
||||||
|
|
||||||
This is a design note for a feature, not an implementation plan — same posture as
|
|
||||||
`SERVICE-ACCOUNTS.md`, and raised for the same reason: `terdut-operator`'s `TerdutTeam`
|
|
||||||
controller (ROADMAP.md Stage 2, a separate repo, no shared code) hit a gap this server has
|
|
||||||
no answer for yet.
|
|
||||||
|
|
||||||
## The problem
|
|
||||||
|
|
||||||
`POST /api/teams` (`handleCreateTeam`, confirmed against `internal/api/teams.go`) lets an
|
|
||||||
instance-scoped service account create a team — it has its own explicit
|
|
||||||
`isInstanceServiceAccount(...)` branch alongside the human-user path, not gated by
|
|
||||||
`AdminOnly`. If that call succeeds server-side but the caller (`TerdutTeam`'s controller)
|
|
||||||
crashes before persisting the resulting team ID locally, a retry's `POST` 409s on the name's
|
|
||||||
unique constraint (confirmed: the `isUniqueViolation` branch in the same handler).
|
|
||||||
|
|
||||||
Recovering from that 409 means looking the team up by name, and nothing today permits that
|
|
||||||
for a service account:
|
|
||||||
|
|
||||||
- `GET /api/teams` (`handleListTeams`) answers "what teams does the *caller* belong to", via
|
|
||||||
a `team_members` join keyed on `userFromContext`'s `caller.ID` — confirmed against source.
|
|
||||||
A service account is never a member of anything, so this always returns empty for one,
|
|
||||||
regardless of what exists.
|
|
||||||
- `GET /api/admin/teams` (`handleAdminListTeams`) is gated by `AdminOnly`, and `AdminOnly`'s
|
|
||||||
actual code (`internal/api/middleware.go`) checks only `userFromContext(...).IsAdmin` — no
|
|
||||||
branch for a service account at all, confirmed against source. This contradicts
|
|
||||||
`SERVICE-ACCOUNTS.md`'s own text, which claims "an instance-scoped [service account
|
|
||||||
satisfies] `AdminOnly` for team-creation/listing purposes" — that claim doesn't match this
|
|
||||||
endpoint's actual, shipped code. (Team *creation* is fine: `handleCreateTeam` isn't behind
|
|
||||||
`AdminOnly` at all, it has its own check. Only the listing half of that sentence is wrong.)
|
|
||||||
|
|
||||||
This is exactly the shape of gap `SERVICE-ACCOUNTS.md`'s own `GET /api/service-accounts?name=`
|
|
||||||
closed for service accounts themselves (confirmed: that endpoint's own comment —
|
|
||||||
"the name lookup is open to any authenticated caller... what lets a service account find its
|
|
||||||
own account on the 403 that follows a second POST"). Teams never got the equivalent, because
|
|
||||||
nothing needed it until an operator started creating them unattended.
|
|
||||||
|
|
||||||
## Goals
|
|
||||||
|
|
||||||
- A service-account-accessible way to look up one team by exact name, mirroring
|
|
||||||
`GET /api/service-accounts?name=` as closely as possible — same shape, same reasoning,
|
|
||||||
same low sensitivity of what it discloses.
|
|
||||||
- No change to today's behavior for an empty/no-name request.
|
|
||||||
|
|
||||||
## Proposed shape
|
|
||||||
|
|
||||||
Extend `GET /api/teams` itself, the same way `handleListServiceAccounts` already branches on
|
|
||||||
a `?name=` query param, rather than adding a new route:
|
|
||||||
|
|
||||||
- `name` unset (today's behavior, unchanged): the caller's own teams, via `team_members`.
|
|
||||||
- `name=<value>` set: look up that one team by exact name — a one-or-zero-length array, not
|
|
||||||
an error on no match, mirroring `GET /api/service-accounts?name=`'s own response shape and
|
|
||||||
status codes exactly. Deliberately **not** gated by `isInstanceServiceAccount` or
|
|
||||||
`AdminOnly`: a human caller who's already a member sees this same information in their own
|
|
||||||
team list regardless, and a non-member learning only that a name is taken — not who's in
|
|
||||||
the team, not any of its data — is the same low-sensitivity disclosure
|
|
||||||
`GET /api/service-accounts?name=` already accepts for service-account names.
|
|
||||||
|
|
||||||
## What this unblocks
|
|
||||||
|
|
||||||
Directly resolves the crash-window gap in `terdut-operator`'s `TerdutTeam` controller: on a
|
|
||||||
409 from `POST /api/teams`, `GET /api/teams?name=<the same name>` — authenticated with the
|
|
||||||
same instance-scoped credential that just got the 409 — finds the id, and the controller
|
|
||||||
proceeds as if its own create had returned it directly. The same adopt-on-409 pattern already
|
|
||||||
proven for service accounts (that repo's `DESIGN.md` §6 point 1, §5's general rule), not a
|
|
||||||
new one.
|
|
||||||
|
|
||||||
## Suggested sequencing
|
|
||||||
|
|
||||||
Land this before `TerdutTeam`'s create path is implemented — the same reasoning
|
|
||||||
`SERVICE-ACCOUNTS.md` gave for its own sequencing: writing that code against today's gap as a
|
|
||||||
"known-temporary workaround" is wasted effort when the fix is this small and this
|
|
||||||
well-precedented.
|
|
||||||
@@ -15,5 +15,5 @@ type: application
|
|||||||
# appVersion and image.tag in values.yaml no longer agree, and that is not an oversight:
|
# appVersion and image.tag in values.yaml no longer agree, and that is not an oversight:
|
||||||
# image.tag stays "latest", which is what a local install actually pulls. appVersion is
|
# image.tag stays "latest", which is what a local install actually pulls. appVersion is
|
||||||
# metadata and drives nothing.
|
# metadata and drives nothing.
|
||||||
version: 0.41.1
|
version: 0.43.0
|
||||||
appVersion: "v0.41.1"
|
appVersion: "v0.43.0"
|
||||||
|
|||||||
@@ -95,12 +95,6 @@ spec:
|
|||||||
value: "{{ .Values.sweeper.staleAfter }}"
|
value: "{{ .Values.sweeper.staleAfter }}"
|
||||||
- name: TERDUT_ARCHIVE_AFTER
|
- name: TERDUT_ARCHIVE_AFTER
|
||||||
value: "{{ .Values.sweeper.archiveAfter }}"
|
value: "{{ .Values.sweeper.archiveAfter }}"
|
||||||
- name: TERDUT_DEADMAN_MATCHERS
|
|
||||||
value: "{{ .Values.deadman.matchers }}"
|
|
||||||
- name: TERDUT_DEADMAN_TIMEOUT
|
|
||||||
value: "{{ .Values.deadman.timeout }}"
|
|
||||||
- name: TERDUT_DEADMAN_SEVERITY
|
|
||||||
value: "{{ .Values.deadman.severity }}"
|
|
||||||
{{- if .Values.notify.ntfyUrl }}
|
{{- if .Values.notify.ntfyUrl }}
|
||||||
- name: TERDUT_NTFY_URL
|
- name: TERDUT_NTFY_URL
|
||||||
value: "{{ .Values.notify.ntfyUrl }}"
|
value: "{{ .Values.notify.ntfyUrl }}"
|
||||||
@@ -122,6 +116,8 @@ spec:
|
|||||||
value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}"
|
value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}"
|
||||||
- name: TERDUT_PASSWORD_LOGIN
|
- name: TERDUT_PASSWORD_LOGIN
|
||||||
value: {{ .Values.passwordLogin | quote }}
|
value: {{ .Values.passwordLogin | quote }}
|
||||||
|
- name: TERDUT_TRUSTED_PROXIES
|
||||||
|
value: {{ .Values.trustedProxies | quote }}
|
||||||
- name: TERDUT_OPERATOR_MODE
|
- name: TERDUT_OPERATOR_MODE
|
||||||
value: {{ .Values.operatorMode | quote }}
|
value: {{ .Values.operatorMode | quote }}
|
||||||
{{- if .Values.oidc.enabled }}
|
{{- if .Values.oidc.enabled }}
|
||||||
|
|||||||
@@ -72,43 +72,10 @@ sweeper:
|
|||||||
# How long a resolved alert stays in the default list before auto-archiving.
|
# How long a resolved alert stays in the default list before auto-archiving.
|
||||||
archiveAfter: 168h
|
archiveAfter: 168h
|
||||||
|
|
||||||
# Alerts treated as dead man's switches: receiving one opens no incident, and
|
# How many reverse proxies in front of the server append to X-Forwarded-For.
|
||||||
# the absence of one does. The Watchdog alert kube-prometheus-stack ships is
|
# The per-address login/sign-up rate limits take the client address that many
|
||||||
# exactly this — an always-firing alert whose only value is something noticing
|
# entries from the right. 0 ignores the header.
|
||||||
# when it stops.
|
trustedProxies: 1
|
||||||
deadman:
|
|
||||||
# Which alerts to treat as heartbeats. ";" separates matchers, "," separates
|
|
||||||
# the label conditions within one, "=" is exact equality. Every matcher must
|
|
||||||
# name an alertname:
|
|
||||||
# alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat
|
|
||||||
# Each distinct label set is watched independently, so two clusters sending
|
|
||||||
# the same alertname are two switches and a live one cannot mask a dead one.
|
|
||||||
matchers: "alertname=Watchdog"
|
|
||||||
# How long a heartbeat may go unheard before its switch is declared dead.
|
|
||||||
#
|
|
||||||
# This must be SHORTER than the Alertmanager repeat_interval of the route
|
|
||||||
# carrying the heartbeat — the opposite of sweeper.staleAfter. The default
|
|
||||||
# repeat_interval of 4h (12h in many setups) makes for a useless dead man's
|
|
||||||
# switch, so give the heartbeat a route of its own:
|
|
||||||
#
|
|
||||||
# - matchers: [ 'alertname = "Watchdog"' ]
|
|
||||||
# receiver: terdut
|
|
||||||
# group_wait: 0s
|
|
||||||
# group_interval: 1m
|
|
||||||
# repeat_interval: 1m
|
|
||||||
#
|
|
||||||
# That delivers every 2m rather than every 1m: a group is only reconsidered
|
|
||||||
# each group_interval, and at exactly one elapsed interval repeat_interval has
|
|
||||||
# not quite passed, so equal values give 2x. Fine against 15m; use
|
|
||||||
# group_interval: 30s if you want a true 1m.
|
|
||||||
#
|
|
||||||
# Set to 0 to disable dead man's switch handling entirely.
|
|
||||||
timeout: 15m
|
|
||||||
# Severity a dead man's switch incident opens at. These incidents have no
|
|
||||||
# member alerts to derive one from, and the heartbeat's own severity label is
|
|
||||||
# meaningless — Watchdog ships as "none". Only "critical" maps to the ntfy
|
|
||||||
# priority that overrides a phone's quiet hours.
|
|
||||||
severity: critical
|
|
||||||
|
|
||||||
notify:
|
notify:
|
||||||
# ntfy server that push notifications are published to, e.g.
|
# ntfy server that push notifications are published to, e.g.
|
||||||
@@ -190,10 +157,8 @@ oidc:
|
|||||||
# Hard ceiling on a session made by a single sign-on login.
|
# Hard ceiling on a session made by a single sign-on login.
|
||||||
sessionMaxAge: 12h
|
sessionMaxAge: 12h
|
||||||
|
|
||||||
# Backups are no longer this chart's business. The SQLite database lived on a PVC
|
# Backups are not this chart's business: Postgres is backed up where it runs,
|
||||||
# beside the app, so it needed a sidecar with a sqlite3 module for k8up to exec a
|
# through a k8up.io/backupcommand pg_dump annotation on the database pod itself.
|
||||||
# dump in; Postgres is backed up where it runs, through a k8up.io/backupcommand
|
|
||||||
# pg_dump annotation on the database pod itself.
|
|
||||||
|
|
||||||
bootstrap:
|
bootstrap:
|
||||||
enabled: true
|
enabled: true
|
||||||
|
|||||||
@@ -39,21 +39,16 @@ func main() {
|
|||||||
RepeatEvery: cfg.NotifyRepeat,
|
RepeatEvery: cfg.NotifyRepeat,
|
||||||
}
|
}
|
||||||
|
|
||||||
// Dead man's switches live per team now. The environment variables are the
|
|
||||||
// defaults a team starts from: every team without a configuration of its
|
|
||||||
// own gets one from them here, and an owner's later edit is never
|
|
||||||
// overwritten by a redeploy.
|
|
||||||
deadman := api.ParseDeadmanConfig(cfg.DeadmanMatchers, cfg.DeadmanTimeout, cfg.DeadmanSeverity)
|
|
||||||
if err := api.SeedDeadmanConfigs(context.Background(), database, deadman); err != nil {
|
|
||||||
log.Fatalf("seed dead man's switch defaults: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// The behaviour knobs move into the database on first start, after which an
|
// The behaviour knobs move into the database on first start, after which an
|
||||||
// administrator owns them and a redeploy leaves them alone.
|
// administrator owns them and a redeploy leaves them alone.
|
||||||
if err := api.SeedSettings(context.Background(), database, cfg); err != nil {
|
if err := api.SeedSettings(context.Background(), database, cfg); err != nil {
|
||||||
log.Fatalf("seed settings: %v", err)
|
log.Fatalf("seed settings: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), database, cfg.OperatorKey); err != nil {
|
||||||
|
log.Fatalf("%v", err)
|
||||||
|
}
|
||||||
|
|
||||||
router := api.NewRouter(database, notify, cfg, version)
|
router := api.NewRouter(database, notify, cfg, version)
|
||||||
|
|
||||||
srv := &http.Server{
|
srv := &http.Server{
|
||||||
|
|||||||
@@ -0,0 +1,21 @@
|
|||||||
|
# Terminal Duty documentation
|
||||||
|
|
||||||
|
The [README](../README.md) is the short tour. These pages hold the detail.
|
||||||
|
|
||||||
|
**Running it**
|
||||||
|
- [Deployment](./deployment.md): Docker, the Helm chart, the database, backups, and the operator.
|
||||||
|
- [Configuration](./configuration.md): environment variables and settings.
|
||||||
|
- [Single sign-on](./single-sign-on.md): OIDC, group mapping, the terminal device flow.
|
||||||
|
|
||||||
|
**Using it**
|
||||||
|
- [The web UI](./web-ui.md): sessions, the Team and Admin tabs.
|
||||||
|
- [Alertmanager configuration](./alertmanager.md): routes, integration keys and webhooks.
|
||||||
|
- [Alerts and incidents](./incidents.md): correlation, lifecycle, on-call assignment, stale-alert expiry.
|
||||||
|
- [Push notifications](./notifications.md): ntfy pages and acknowledging from them.
|
||||||
|
- [Escalation](./escalation.md): ladders, repeats and the fallback topic.
|
||||||
|
- [Dead man's switches](./dead-mans-switch.md): noticing that alerts stopped arriving.
|
||||||
|
|
||||||
|
**Integrating and contributing**
|
||||||
|
- [API reference](./api.md): every endpoint, authentication and error shape.
|
||||||
|
- [Service accounts](../SERVICE-ACCOUNTS.md): non-human credentials for automation.
|
||||||
|
- [Development and releasing](./development.md): tests, the CI gate, the release pipeline.
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
# Alertmanager configuration
|
||||||
|
|
||||||
|
_Pointing Alertmanager at the server._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Alerts arrive on a team's **integration key**, which says both that the sender
|
||||||
|
may post and which team the alerts belong to. Mint one as an owner of the team:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl -X POST https://terdut.example.com/api/teams/1/integrations \
|
||||||
|
-H "Authorization: Bearer $TERDUT_API_KEY" \
|
||||||
|
-H 'Content-Type: application/json' \
|
||||||
|
-d '{"name":"prod alertmanager"}'
|
||||||
|
```
|
||||||
|
|
||||||
|
The response carries the key and the full URL **once**; only a SHA-256 hash is
|
||||||
|
stored. Put it in your `alertmanager.yml`:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
receivers:
|
||||||
|
- name: terdut
|
||||||
|
webhook_configs:
|
||||||
|
- url: http://terdut-server:8080/api/integrations/<key>/alertmanager
|
||||||
|
send_resolved: true
|
||||||
|
|
||||||
|
route:
|
||||||
|
receiver: terdut
|
||||||
|
```
|
||||||
|
|
||||||
|
The whole URL is a credential, so treat it like one. Alertmanager 0.26 and
|
||||||
|
later can read it from a file with `url_file:` instead, which keeps it out of
|
||||||
|
your configuration repository:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
- url_file: /etc/alertmanager/secrets/terdut-webhook-url/url
|
||||||
|
send_resolved: true
|
||||||
|
```
|
||||||
|
|
||||||
|
The webhook endpoint requires no authentication.
|
||||||
|
|
||||||
|
If you use the [dead man's switch](./dead-mans-switch.md) — and the default configuration does — give
|
||||||
|
the heartbeat a route of its own, because the deadline is only as tight as the interval feeding it:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
route:
|
||||||
|
receiver: terdut
|
||||||
|
repeat_interval: 4h
|
||||||
|
routes:
|
||||||
|
- matchers: [ 'alertname = "Watchdog"' ]
|
||||||
|
receiver: terdut
|
||||||
|
group_wait: 0s
|
||||||
|
group_interval: 1m
|
||||||
|
repeat_interval: 1m
|
||||||
|
```
|
||||||
|
|
||||||
|
That delivers a heartbeat every **2 minutes**, not every minute. Alertmanager only reconsiders a
|
||||||
|
group every `group_interval`, and at exactly one elapsed interval `repeat_interval` has not *quite*
|
||||||
|
passed, so the send slips to the next tick — equal values give 2×. Two minutes against the 15 minute
|
||||||
|
default is seven heartbeats per window, which is the point; use `group_interval: 30s` if you want
|
||||||
|
the numbers to mean what they say.
|
||||||
|
|
||||||
|
kube-prometheus-stack users get the `Watchdog` alert (`expr: vector(1)`) for free; it just needs
|
||||||
|
routing to terdut rather than to `null`.
|
||||||
@@ -0,0 +1,436 @@
|
|||||||
|
# API reference
|
||||||
|
|
||||||
|
_The REST API._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
## Authentication
|
||||||
|
|
||||||
|
All endpoints except `/api/bootstrap`, `/api/integrations/{key}/alertmanager`,
|
||||||
|
`/api/notify/ack/{token}`, `/api/login`, `/api/logout`, `/api/auth/config`,
|
||||||
|
`/api/version`, `/api/oidc/login`, `/api/oidc/callback`, `/api/oidc/device` and
|
||||||
|
`/api/oidc/device/token` require either an API key:
|
||||||
|
|
||||||
|
```
|
||||||
|
Authorization: Bearer <api-key>
|
||||||
|
```
|
||||||
|
|
||||||
|
or the web UI's session cookie. A request that carries an `Authorization` header
|
||||||
|
is judged on that header alone.
|
||||||
|
|
||||||
|
Two kinds of user exist. An **administrator** manages accounts: creating and
|
||||||
|
deleting users, setting anybody's password, minting keys for anybody, and
|
||||||
|
granting the flag itself. Everybody else works incidents — acknowledging,
|
||||||
|
assigning, snoozing, resolving, noting — and manages their own account and
|
||||||
|
nobody else's. An API key carries exactly the rights of the user it belongs to.
|
||||||
|
|
||||||
|
A third principal, the **service account**, exists for automation (a
|
||||||
|
Kubernetes operator, most likely) that needs to manage teams, escalation
|
||||||
|
policies, dead man's switches and integrations without impersonating a human.
|
||||||
|
It is not a user — it never signs in, never appears in a team's member list,
|
||||||
|
and never holds the administrator flag — and its key is prefixed `tdsa_` so it
|
||||||
|
reads as one at a glance in a log line. See [Service accounts](#service-accounts).
|
||||||
|
|
||||||
|
**Getting an account.** The first one comes from `/api/bootstrap`. After that
|
||||||
|
it depends on `signup_mode`, an administrator setting:
|
||||||
|
|
||||||
|
- `invite_only` (the default) — a team owner mints a link with
|
||||||
|
`POST /api/teams/{teamID}/invites`, and the person who opens it picks a
|
||||||
|
username and password and lands in that team with the role the link carries.
|
||||||
|
Links are single-use unless told otherwise, expire after seven days, and can
|
||||||
|
be revoked before that.
|
||||||
|
- `open` — anybody who can reach the server can create an account, and must
|
||||||
|
name a team, which they then own.
|
||||||
|
|
||||||
|
Invites are **links, not email**: this server has no SMTP, and adding it to send
|
||||||
|
one message would be a subsystem to run, secure and monitor. Send the link
|
||||||
|
however you already talk to the person.
|
||||||
|
|
||||||
|
A domain-restricted third mode was considered and dropped: with no email there
|
||||||
|
is nothing to verify an address against, so it would only check the domain of a
|
||||||
|
string somebody typed.
|
||||||
|
|
||||||
|
The first user, from `/api/bootstrap`, is an administrator. Users created
|
||||||
|
afterwards are not, until an administrator says so. An install always keeps at
|
||||||
|
least one: the last administrator can be neither deleted nor demoted, and
|
||||||
|
nobody can delete or demote themselves.
|
||||||
|
|
||||||
|
Endpoints that require the flag answer `403` with
|
||||||
|
`{"error":"administrator access required"}`.
|
||||||
|
|
||||||
|
**Teams** are the unit of tenancy, and are a separate axis from the administrator
|
||||||
|
flag. A team owns its incidents, alerts, schedule and integrations, and a user
|
||||||
|
sees exactly the teams they belong to. Within a team an **owner** configures it
|
||||||
|
(schedule, integrations, membership) and a **member** works its incidents.
|
||||||
|
|
||||||
|
An administrator crosses that line in one direction only. They **configure any
|
||||||
|
team** without being in it — every owner-only endpoint accepts the flag, because
|
||||||
|
otherwise a team whose last owner left could never be repaired. They do **not
|
||||||
|
read any team**: the queue, the alerts and the incidents are filtered by real
|
||||||
|
membership, so an administrator sees a team's work only by joining it, which is
|
||||||
|
a membership change and shows up as one. Administration is about accounts and
|
||||||
|
the shape of a team, not about reading other people's incidents.
|
||||||
|
|
||||||
|
Anything belonging to a team you are not in answers `404`, not `403`: whether an
|
||||||
|
incident exists is itself something only its team should learn.
|
||||||
|
|
||||||
|
**Operator mode** (`TERDUT_OPERATOR_MODE`, see [Configuration](./configuration.md#configuration))
|
||||||
|
declares this install gitops-managed. When it is on, a session or a user's own
|
||||||
|
API key gets `403 {"error": "...", "reason": "operator_managed"}` on every
|
||||||
|
write this page marks **owner**-gated under Teams below (creating, renaming
|
||||||
|
or deleting a team; its OIDC group binding; its escalation ladder; its dead
|
||||||
|
man's switches; its integrations) — a service account's writes are unaffected.
|
||||||
|
Team membership and invites are deliberately excluded: they are never
|
||||||
|
gitops-managed, in operator mode or out of it. `GET /api/auth/config` reports
|
||||||
|
`operator_mode` so a client can grey those sections out before a write is ever
|
||||||
|
attempted.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/auth/config` | How to sign in: `{"password_login", "oidc": {"enabled","name"}, "device_login", "operator_mode"}`. No session needed |
|
||||||
|
| `GET` | `/api/version` | `{"version"}` — this build's version string. No session needed, the same as `/healthz` |
|
||||||
|
| `POST` | `/api/login` | `{"username","password"}` → sets the session cookie, returns `{user, has_password}`. `429` after too many failures; `403` when `TERDUT_PASSWORD_LOGIN=false` |
|
||||||
|
| `GET` | `/api/oidc/login` | Starts a single sign-on sign-in: redirects the browser to the provider. `?next=/path` is where to land afterwards; only a path on this server is honoured. Only exists when SSO is configured |
|
||||||
|
| `POST` | `/api/oidc/device` | Starts a device login: returns `{device_code, user_code, verification_url, interval, expires_in}`. Only exists when SSO is configured |
|
||||||
|
| `POST` | `/api/oidc/device/token` | `{"device_code"}` → `202 {"status":"pending"}`, then `200` with the session cookie once approved (once only). `410` with `{"error":"expired"}` or `{"error":"denied"}`; `429 {"error":"slow_down"}` if polled faster than `interval` |
|
||||||
|
| `POST` | `/api/oidc/device/approve` | **session** — `{"user_code"}`. Approves a pending device login as the caller. `403` for an API key; `404` for an unknown, expired or already decided code |
|
||||||
|
| `POST` | `/api/oidc/device/deny` | **session** — `{"user_code"}`. Refuses it |
|
||||||
|
| `GET` | `/api/oidc/callback` | Where the provider sends the browser back. Sets the session cookie and redirects to `/`, or to `/?sso_error=<code>` — one of `denied`, `expired`, `failed`, `unavailable`, `not_allowed`, `no_email`, `email_conflict`, `disabled`, `not_bootstrapped` (no user exists on this install yet — sign in again once something has called `/api/bootstrap`) |
|
||||||
|
| `POST` | `/api/logout` | Ends the session and clears the cookie |
|
||||||
|
| `GET` | `/api/me` | The caller: `{user, has_password}` |
|
||||||
|
|
||||||
|
## Users
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
**admin** marks an endpoint that requires the administrator flag; **self or
|
||||||
|
admin** marks one you may use on your own account and an administrator may use
|
||||||
|
on anybody's.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/signup` | — | Whether sign-up is open, and whether `?invite=` is usable. No session needed: the caller has no account yet |
|
||||||
|
| `POST` | `/api/signup` | — | Create an account `{"username","email","password","invite"?,"team_name"?}` and sign in. `403` without a usable invite when the mode is invite-only |
|
||||||
|
| `POST` | `/api/bootstrap` | — | Create first user + API key `{"username","email","password"?}` (only works on empty DB). The user is an administrator |
|
||||||
|
| `GET` | `/api/users` | any | List users. Open to everybody: the queue's assignment control and the schedule both have to name people |
|
||||||
|
| `GET` | `/api/users/{id}/teams` | self or admin | The teams that user is in, each with their role. `/api/teams` is always about the caller; this one answers it about somebody else, for the admin page's per-user view. `404` for a user who does not exist, so "no teams" and "no such person" are distinguishable |
|
||||||
|
| `POST` | `/api/users` | **admin** | Create user `{"username","email"}`. Not an administrator |
|
||||||
|
| `DELETE` | `/api/users/{id}` | **admin** | Delete user (cascades to keys). `409` for yourself or the last administrator |
|
||||||
|
| `PUT` | `/api/users/{id}/admin` | **admin** | Grant or revoke the administrator flag `{"is_admin"}`. `409` for yourself, the last administrator, or an administrator granted by single sign-on |
|
||||||
|
| `PUT` | `/api/users/{id}/disabled` | **admin** | Take an account out of use, or put it back `{"disabled"}`. `409` for yourself or the last administrator |
|
||||||
|
| `PUT` | `/api/users/{id}/notify` | self or admin | Set push notification target `{"ntfy_topic"}` — empty string clears it |
|
||||||
|
| `PUT` | `/api/users/{id}/password` | self or admin | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions |
|
||||||
|
| `POST` | `/api/users/{id}/api-keys` | self or admin | Issue API key `{"name"}` — key shown once |
|
||||||
|
| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | self or admin | Revoke API key |
|
||||||
|
|
||||||
|
## Administration
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/admin/teams` | **admin** | Every team on the server, with its member and open-incident counts. `/api/teams` answers "what am I in"; this answers "what is there" |
|
||||||
|
| `GET` | `/api/admin/teams/{teamID}` | **admin** | One team and who is in it: `{"team", "members"}`. `404` for a team that does not exist. `GET /api/teams/{teamID}/members` is **member**-only and still `404`s an administrator from outside the team — reading a team's shape and reading its work are different questions, so they are different endpoints |
|
||||||
|
| `GET` | `/api/admin/settings` | **admin** | The editable settings with their bounds, plus the environment-configured ones, read-only. Never credentials |
|
||||||
|
| `PUT` | `/api/admin/settings` | **admin** | Change one or more `{"key": seconds}`, or `{"signup_mode": "open"\|"invite_only"}`. `400` for an unknown key or a value outside its bounds |
|
||||||
|
|
||||||
|
## Service accounts
|
||||||
|
|
||||||
|
A service account is a scoped, non-human credential for automation — not a
|
||||||
|
`users` row, so it never signs in, is never a team member, and never carries
|
||||||
|
the administrator flag. Two scopes:
|
||||||
|
|
||||||
|
- **instance** — the same reach system administration has over teams: create
|
||||||
|
one, and mint a **team**-scoped account against any of them. There is no
|
||||||
|
cap on how many instance-scoped accounts exist, but ordinarily there is one,
|
||||||
|
belonging to whatever is provisioning this install end to end.
|
||||||
|
- **team** — owner-equivalent for that one team, and nothing else: every
|
||||||
|
**owner**-gated endpoint under [Teams](#teams), membership and invites
|
||||||
|
included. Nothing narrower is enforced server-side; what actually keeps
|
||||||
|
membership out of automation's hands is that no operator built against this
|
||||||
|
scope should ever call those two endpoints — see
|
||||||
|
[operator mode](#authentication) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)'s note on this.
|
||||||
|
|
||||||
|
A key is shown once, at creation or rotation, and only its hash is stored —
|
||||||
|
the same handling as a user's API key. Losing it means minting a new one;
|
||||||
|
there is no way to recover a raw key from the server.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/service-accounts` | **admin** | Every service account. Pass `?name=` instead to look one up by its exact name — open to **any** authenticated caller (human or service account), since it returns no key material and is how an account finds its own id |
|
||||||
|
| `POST` | `/api/service-accounts` | owner\* | Create one and mint its first key `{"name","scope","team_id"?}` (`team_id` required for `scope:"team"`, absent for `scope:"instance"`). Returns `{"service_account", "key"}` — `key.key` shown once |
|
||||||
|
| `POST` | `/api/service-accounts/{id}/keys` | owner\* | Mint an additional key `{"name"}` — rotation without recreating the account. Shown once |
|
||||||
|
| `DELETE` | `/api/service-accounts/{id}/keys/{keyID}` | owner\* | Revoke one key |
|
||||||
|
|
||||||
|
\* For an **instance**-scoped account: a system administrator only. For a
|
||||||
|
**team**-scoped account: a system administrator, that team's own human owner,
|
||||||
|
an instance-scoped service account (minting a narrower credential for a team
|
||||||
|
it just created), or — for the two key endpoints only — the account rotating
|
||||||
|
or revoking its own key, which is not a privilege escalation, the same
|
||||||
|
reasoning a user's own API keys rest on.
|
||||||
|
|
||||||
|
## Alert ingestion
|
||||||
|
|
||||||
|
Alerts arrive on a team's integration key. The key is both the credential and the
|
||||||
|
routing: it says that the sender may post, and which team the alerts belong to.
|
||||||
|
Create one with `POST /api/teams/{teamID}/integrations`, which returns the key
|
||||||
|
and the full URL once and stores only a SHA-256 hash.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `POST` | `/api/integrations/{key}/alertmanager` | Alertmanager v4 webhook receiver for the key's team. `401` for an unknown key |
|
||||||
|
|
||||||
|
This is the only way in. The pre-teams `POST /api/alertmanager/webhook` took no
|
||||||
|
credential at all — anything able to reach the port could open an incident —
|
||||||
|
and was removed in v0.13.0 once senders had moved onto keys.
|
||||||
|
|
||||||
|
## Teams
|
||||||
|
|
||||||
|
**owner** below means an owner of that team, a system administrator (who
|
||||||
|
passes every one of these without being a member), or that team's own
|
||||||
|
team-scoped [service account](#service-accounts) — including membership and
|
||||||
|
invites, technically, though no automation this scope was designed for
|
||||||
|
(a Kubernetes operator's CRDs, see [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)) ever models team
|
||||||
|
membership or would call those two. See [Authentication](#authentication).
|
||||||
|
**member** means membership and nothing else: an administrator who is not in
|
||||||
|
the team gets the same `404` as anybody else.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/teams` | any | The caller's own teams, each with their role |
|
||||||
|
| `POST` | `/api/teams` | any | Create a team `{"name"}`; a human creator becomes its first owner. An instance-scoped [service account](#service-accounts) may also create one, and it gets no owner at all — expected for a team an operator is about to hand a team-scoped credential to, not an orphaned team a human made |
|
||||||
|
| `PUT` | `/api/teams/{teamID}` | **owner** | Rename it `{"name"}`. `409` if the name is taken |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}` | **owner** | Delete a team and everything under it. `409` while it has open incidents |
|
||||||
|
| `GET` | `/api/teams/{teamID}/members` | member | Who is in the team, with `status` (`oncall` if the rota has them today, `unpageable` when a page to them would go nowhere — even if they are on call — else `reachable`), `on_call`, `next_shift` (first rota day after today), `pageable` and `problem` (`has no ntfy topic` / `account is disabled`; never the topic itself) and `last_active_at` (their newest session or API-key use). Every member sees the same list |
|
||||||
|
| `POST` | `/api/teams/{teamID}/members` | **owner** | Add a member, or change their role `{"user_id","role"}`. `409` when it would demote the last owner, or the membership is managed by single sign-on |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/members/{userID}` | **owner** | Remove a member. `409` for the last owner, or a membership managed by single sign-on |
|
||||||
|
| `GET` | `/api/teams/{teamID}/oidc-groups` | member | Which groups control this team's membership: `{"member_group","owner_group"}`. An empty string means no group grants that role here |
|
||||||
|
| `PUT` | `/api/teams/{teamID}/oidc-groups` | **owner** | Set them. An empty string clears a binding |
|
||||||
|
| `GET` | `/api/teams/{teamID}/integrations` | member | List integrations. Never returns keys. Each carries `status` (`active` if its key posted within 24h, `quiet` if it has but not lately, `never`), `last_used_at` (last webhook, usable or not), `last_alert_at` (when an alert last arrived on it) and `alerts_24h` (distinct alerts it refreshed in the last day). Alerts delivered before the source was recorded (migration 010) have none, so the last two fill in as Alertmanager re-sends them |
|
||||||
|
| `PATCH` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Rename `{"name"}`. The key does not change |
|
||||||
|
| `POST` | `/api/teams/{teamID}/integrations` | **owner** | Mint an integration `{"name","kind"}` — key and URL shown once |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Revoke an integration. Alerts it delivered stay, unattributed |
|
||||||
|
| `GET` | `/api/teams/{teamID}/invites` | **owner** | The team's invite links, with their uses and expiry. Never the tokens |
|
||||||
|
| `POST` | `/api/teams/{teamID}/invites` | **owner** | Mint one `{"role","max_uses"}` — the full URL is returned once |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/invites/{inviteID}` | **owner** | Revoke a link before it expires |
|
||||||
|
| `GET` | `/api/teams/{teamID}/escalation` | member | The team's [escalation ladder](./escalation.md#escalation) `{repeat_count, fallback_topic, levels[], last_escalated_at?, last_escalated_incident_id?}`. Empty levels means the team has none. Each level also carries `status` (`ready`, `escalating` when an unanswered incident has climbed to it, `unreachable` when nobody on it could be woken), `waiting` (ids of the open incidents on it) and, per target, `username` (who it means today — the person on call, for a rota target), `reachable` and `problem`. The extra fields are output only; `PUT` takes the plain shape |
|
||||||
|
| `PUT` | `/api/teams/{teamID}/escalation` | **owner** | Replace it wholesale. `400` for a level with no targets or no timeout — a rung that pages nobody is a silence with a number on it |
|
||||||
|
| `GET` | `/api/teams/{teamID}/deadman/switches` | member | The team's [dead man's switches](./dead-mans-switch.md), each `{id, name, matcher, timeout_seconds, severity, status, last_heartbeat_at, last_triggered_at, open_incident_id, sources[]}`. `status` is `healthy`, `dead` or `dormant`; `sources` has one entry per heartbeat fingerprint. Empty when the team watches nothing |
|
||||||
|
| `POST` | `/api/teams/{teamID}/deadman/switches` | **owner** | Add one: `{name?, matcher, timeout_seconds, severity?}`. `400` when the matcher names no `alertname` or holds several, or the timeout is not positive — a switch that silently watches nothing is the failure this feature exists to prevent |
|
||||||
|
| `PUT` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Replace one in place, same body and validation as create. Its id is unchanged — for an automated caller reconciling a spec change, unlike delete-and-recreate |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Stop watching. An incident it opened stays open. `404` for a switch of another team |
|
||||||
|
|
||||||
|
## Notifications
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `POST` | `/api/notify/ack/{token}` | Acknowledge an incident from a push notification's Acknowledge button. No auth: the token in the path is the credential — one incident, one action, 24 hours, idempotent. Must stay publicly reachable |
|
||||||
|
|
||||||
|
## Incidents
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/incidents` | List incidents. Filters: `?status=triggered\|acknowledged\|resolved`, `?severity=`, `?assigned_to=<user id>`, `?archived=true`, `?snoozed=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?sort=severity`, `?cluster=<value of the cluster group label>`, `?limit=` (default 50, max 500) |
|
||||||
|
| `GET` | `/api/incidents/clusters` | The distinct `cluster` values on the caller's incidents from the last 90 days, sorted (`?team_id=` narrows it). An empty array when nothing carries the label |
|
||||||
|
| `GET` | `/api/incidents/{id}` | Get single incident, with its alerts inline |
|
||||||
|
| `GET` | `/api/incidents/{id}/alerts` | Alerts under this incident |
|
||||||
|
| `GET` | `/api/incidents/{id}/timeline` | Full event history, chronological |
|
||||||
|
| `POST` | `/api/incidents/{id}/acknowledge` | Acknowledge (stamps authed user + time) |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/acknowledge` | Clear acknowledgement, back to `triggered` |
|
||||||
|
| `POST` | `/api/incidents/{id}/resolve` | Close by hand — **terminal**, see above |
|
||||||
|
| `POST` | `/api/incidents/{id}/assign` | Reassign `{"user_id"}` |
|
||||||
|
| `POST` | `/api/incidents/{id}/snooze` | Hide until `{"until": RFC3339}` or `{"duration": "2h"}` |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/snooze` | Un-snooze |
|
||||||
|
| `POST` | `/api/incidents/{id}/archive` | Archive (hides from the default list) |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/archive` | Un-archive |
|
||||||
|
| `POST` | `/api/incidents/{id}/notes` | Add a note `{"content"}` |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/notes/{eventID}` | Delete own note |
|
||||||
|
|
||||||
|
With no `?status=` filter, `GET /api/incidents` returns **open** incidents only —
|
||||||
|
the queue an on-call person wants. Currently snoozed and archived incidents are
|
||||||
|
excluded unless asked for. Actions that only make sense on an open incident
|
||||||
|
return `409` once it is resolved.
|
||||||
|
|
||||||
|
Notes are ordinary timeline events of type `note`; only they are deletable, and
|
||||||
|
only by their author. The rest of the timeline is a record of what happened.
|
||||||
|
|
||||||
|
### The incident object
|
||||||
|
|
||||||
|
| Field | Type | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `id` | integer | Server-assigned |
|
||||||
|
| `group_key` | string | Alertmanager's `groupKey` — opaque, treat as an identifier |
|
||||||
|
| `title` | string | Rendered from `groupLabels` |
|
||||||
|
| `group_labels` | object | String→string, as sent by Alertmanager |
|
||||||
|
| `status` | string | `"triggered"`, `"acknowledged"` or `"resolved"` |
|
||||||
|
| `severity` | string | *optional* — high-water mark across the incident's alerts; never lowered |
|
||||||
|
| `triggered_at` | timestamp | When the incident opened |
|
||||||
|
| `acknowledged_by_id` / `acknowledged_by` / `acknowledged_at` | | *optional* — user id, username, time |
|
||||||
|
| `assigned_to_id` / `assigned_to` | | *optional* — user id, username |
|
||||||
|
| `snoozed_until` | timestamp | *optional* — a value in the past reads as not snoozed |
|
||||||
|
| `resolved_at` | timestamp | *optional* |
|
||||||
|
| `resolution_source` | string | *optional* — `"alerts"`, `"manual"` or `"recovered"` |
|
||||||
|
| `archived_at` | timestamp | *optional* |
|
||||||
|
| `alerts` | array | Only on `GET /api/incidents/{id}` |
|
||||||
|
|
||||||
|
Treat `resolution_source` as an open set, as with the alert field of the same
|
||||||
|
name: degrade unknown values to "resolved, reason unknown".
|
||||||
|
|
||||||
|
### The timeline event object
|
||||||
|
|
||||||
|
| Field | Type | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `id` | integer | |
|
||||||
|
| `incident_id` | integer | |
|
||||||
|
| `type` | string | See below — treat as an open set |
|
||||||
|
| `user_id` / `username` | | *optional* — absent when the server acted rather than a person |
|
||||||
|
| `alert_id` | integer | *optional* — the alert an `alert_added` / `alert_resolved` event refers to |
|
||||||
|
| `detail` | string | *optional* — the note body, the snooze deadline, etc. |
|
||||||
|
| `created_at` | timestamp | |
|
||||||
|
|
||||||
|
Types written today: `triggered`, `alert_added`, `alert_resolved`,
|
||||||
|
`acknowledged`, `unacknowledged`, `assigned`, `archived`, `unarchived`, `snoozed`,
|
||||||
|
`unsnoozed`, `resolved`, `note`, `notified`, `notify_failed`, `deadman_silent`. On an
|
||||||
|
`assigned` event `user_id` is the **assignee**, not the actor; the actor is in
|
||||||
|
`actor_user_id`/`actor_username` or `actor_service_account_id`/`actor_service_account_name`
|
||||||
|
(absent on assignments made before they were recorded). New types may be added; render
|
||||||
|
unknown ones generically rather than dropping them.
|
||||||
|
|
||||||
|
On `notified` and `notify_failed`, `detail` carries the notification kind
|
||||||
|
(`triggered` | `reminder` | `resolved`), and on a failure the reason after it.
|
||||||
|
`user_id` is who was paged — absent means the page went to the shared fallback
|
||||||
|
topic and so belongs to nobody. The topic itself is never written to the
|
||||||
|
timeline: it is a shared secret with the ntfy server, and every API key can read
|
||||||
|
this.
|
||||||
|
|
||||||
|
## Alerts
|
||||||
|
|
||||||
|
Alerts are read-only. Everything a person does happens on the incident.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/alerts` | List alerts. Filters: `?status=firing\|resolved`, `?name=`, `?incident_id=`, `?archived=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?limit=` (default 50, max 500) |
|
||||||
|
| `GET` | `/api/alerts/{id}` | Get single alert |
|
||||||
|
|
||||||
|
Archived alerts are hidden from `GET /api/alerts` unless `?archived=true` is
|
||||||
|
passed; alert archiving is automatic housekeeping by the sweeper, not a user
|
||||||
|
action. Resolved alerts carry `resolution_source`: `"alertmanager"` for a real
|
||||||
|
resolved webhook, `"expiry"` when the sweeper inferred it (see
|
||||||
|
[Stale alert expiry](./incidents.md#stale-alert-expiry)), `"deadman"` for a heartbeat declared
|
||||||
|
dead (see [Dead man's switch](./dead-mans-switch.md)).
|
||||||
|
|
||||||
|
### The alert object
|
||||||
|
|
||||||
|
Returned by `GET /api/alerts` (as an array) and `GET /api/alerts/{id}`.
|
||||||
|
Timestamps are RFC 3339 in UTC. Fields marked *optional* are omitted entirely
|
||||||
|
when unset, so clients must treat them as nullable.
|
||||||
|
|
||||||
|
| Field | Type | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `id` | integer | Server-assigned; stable for the life of the row |
|
||||||
|
| `fingerprint` | string | Alertmanager's fingerprint — the upsert key |
|
||||||
|
| `name` | string | From the `alertname` label |
|
||||||
|
| `status` | string | `"firing"` or `"resolved"` |
|
||||||
|
| `labels` | object | String→string, as sent by Alertmanager |
|
||||||
|
| `annotations` | object | String→string, as sent by Alertmanager |
|
||||||
|
| `starts_at` | timestamp | When the alert instance began, **per Prometheus** |
|
||||||
|
| `ends_at` | timestamp | *optional* — absent while no end is known |
|
||||||
|
| `generator_url` | string | Link back to the originating Prometheus |
|
||||||
|
| `received_at` | timestamp | When the server last accepted a webhook for this alert — see below |
|
||||||
|
| `incident_id` | integer | *optional* — the most recent incident this alert belongs to |
|
||||||
|
| `resolution_source` | string | *optional* — `"alertmanager"`, `"expiry"` or `"deadman"` |
|
||||||
|
| `archived_at` | timestamp | *optional* — set while archived |
|
||||||
|
|
||||||
|
#### `received_at` is a liveness heartbeat
|
||||||
|
|
||||||
|
`starts_at` comes from Prometheus and **never changes** for the lifetime of an
|
||||||
|
alert instance. It says when the problem began, not whether it is still
|
||||||
|
happening — an alert that started twelve days ago looks identical whether
|
||||||
|
Alertmanager refreshed it a minute ago or went silent a week ago.
|
||||||
|
|
||||||
|
`received_at` is the field that answers "is this still live". It is set to the
|
||||||
|
server's clock on **every accepted webhook** for that fingerprint, including the
|
||||||
|
unchanged firing notifications Alertmanager re-sends every `repeat_interval`.
|
||||||
|
Clients may rely on this:
|
||||||
|
|
||||||
|
- **A firing alert whose `received_at` is advancing is still being refreshed.**
|
||||||
|
Stale-dating it against `repeat_interval` is a valid liveness check, and it is
|
||||||
|
what the built-in sweeper does (see
|
||||||
|
[Stale alert expiry](./incidents.md#stale-alert-expiry)).
|
||||||
|
- **`received_at` tracks accepted payloads, not delivery attempts.** A retry
|
||||||
|
that describes an older instance than the stored one is discarded, and a
|
||||||
|
discarded payload does not move `received_at`.
|
||||||
|
- **It stops advancing once the alert resolves,** because Alertmanager stops
|
||||||
|
re-sending. On an alert resolved by the sweeper
|
||||||
|
(`"resolution_source": "expiry"`) it therefore marks the last time
|
||||||
|
Alertmanager was actually heard from, which is earlier than `ends_at`.
|
||||||
|
|
||||||
|
`GET /api/alerts` is ordered by `received_at` descending — most recently
|
||||||
|
refreshed first — and the `?from=` / `?to=` filters on both the alert and stats
|
||||||
|
endpoints select on `received_at`, not `starts_at`.
|
||||||
|
|
||||||
|
#### `resolution_source` says how much to trust `ends_at`
|
||||||
|
|
||||||
|
An alert can leave the firing state two ways, and `resolution_source` records
|
||||||
|
which happened. Clients may rely on this:
|
||||||
|
|
||||||
|
- **Absent while firing.** It is set only on resolve, and a re-fire under the
|
||||||
|
same fingerprint clears it again, so its presence always agrees with
|
||||||
|
`"status": "resolved"`.
|
||||||
|
- **`"alertmanager"` — a real resolved webhook arrived.** `ends_at` is the end
|
||||||
|
time Alertmanager reported. It is an observed value and can be displayed as
|
||||||
|
fact.
|
||||||
|
- **`"expiry"` — the sweeper inferred the resolve** because Alertmanager stopped
|
||||||
|
refreshing the alert (see [Stale alert expiry](./incidents.md#stale-alert-expiry)). Nothing
|
||||||
|
ever reported an end, so **`ends_at` is approximate**: it is either the stale
|
||||||
|
`endsAt` watermark from the last notification, or — when that notification
|
||||||
|
carried none — the time the sweep ran, which lags the last real contact by up
|
||||||
|
to `TERDUT_STALE_AFTER` plus a sweep interval. Treat it as "no later than",
|
||||||
|
not as when the problem stopped.
|
||||||
|
|
||||||
|
On these alerts `received_at` is the more truthful signal: it marks the last
|
||||||
|
time Alertmanager was actually heard from. Surfacing the distinction is
|
||||||
|
worthwhile, since `"expiry"` can also mean the alert is still firing and the
|
||||||
|
notification path broke.
|
||||||
|
|
||||||
|
- **`"deadman"` — a heartbeat was declared dead** (see
|
||||||
|
[Dead man's switch](./dead-mans-switch.md)). Like `"expiry"`, an inference from
|
||||||
|
silence rather than an observed end, so `ends_at` is approximate — but a much
|
||||||
|
tighter one, bounded by the switch's timeout. It is also the one resolution
|
||||||
|
a re-fire under the same `starts_at` can undo, since the switch coming back is
|
||||||
|
exactly the evidence that the inference was wrong.
|
||||||
|
|
||||||
|
Treat the value as an open set and tolerate ones you do not recognise — new
|
||||||
|
sources may be added, and unknown values should degrade to "resolved, reason
|
||||||
|
unknown" rather than being rejected.
|
||||||
|
|
||||||
|
## On-call schedule
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
Each team keeps its own rota, so two teams can have two different people on call
|
||||||
|
on the same day. The person taking a shift has to be in the team — paging
|
||||||
|
somebody who cannot open the incident is worse than paging nobody.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `POST` | `/api/teams/{teamID}/schedule` | **owner** | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing |
|
||||||
|
| `GET` | `/api/teams/{teamID}/schedule` | member | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/schedule/{id}` | **owner** | Remove schedule entry |
|
||||||
|
| `GET` | `/api/schedule/current` | any | Who is on call today (UTC) in **every** team the caller is in — one entry per team, `[]` when nobody anywhere |
|
||||||
|
|
||||||
|
## Statistics
|
||||||
|
|
||||||
|
Every figure counts the caller's own teams only: a report that counted other
|
||||||
|
teams' incidents would leak their volume, and their alert names through the
|
||||||
|
top-alerts list, and would not be a number about the reader's work anyway.
|
||||||
|
|
||||||
|
All stat endpoints accept optional `?from=YYYY-MM-DD` and `?to=YYYY-MM-DD`, and exclude archived rows to match the default list views. Alert stats filter on `received_at`; incident stats filter on `triggered_at`.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/stats/incidents` | `{total, triggered, acknowledged, resolved, mtta_seconds, mttr_seconds}` |
|
||||||
|
| `GET` | `/api/stats/alerts` | `{total, firing, resolved}` counts |
|
||||||
|
| `GET` | `/api/stats/alerts/top` | Most frequent alert names. `?limit=` (default 10, max 100) |
|
||||||
|
| `GET` | `/api/stats/alerts/by-hour` | Count per hour-of-day (UTC), all 24 slots returned |
|
||||||
|
| `GET` | `/api/stats/alerts/by-day` | Count per day-of-week, all 7 slots with names returned |
|
||||||
|
|
||||||
|
`mtta_seconds` (time to acknowledge) and `mttr_seconds` (time to resolve) are
|
||||||
|
averages over incidents that have actually been acknowledged or resolved, and are
|
||||||
|
**null** until there are any — null means "no data", not zero.
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# Configuration
|
||||||
|
|
||||||
|
_Environment variables and settings._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Two kinds of setting, split by who changes them and how often.
|
||||||
|
|
||||||
|
**Where the server is plugged in** stays in the environment: the listen address,
|
||||||
|
the database DSN, the ntfy URL and token, the public URL. They are needed before
|
||||||
|
the database is open, and two of them are credentials.
|
||||||
|
|
||||||
|
**How the server behaves** lives in the database and is edited by an
|
||||||
|
administrator in the web UI or through `PUT /api/admin/settings`, taking effect
|
||||||
|
on the next sweep rather than at the next restart. The variables below marked
|
||||||
|
**seed** are the value each of those starts from: written once, on first start,
|
||||||
|
and never overwritten afterwards — a redeploy cannot put a chart's default back
|
||||||
|
over an administrator's edit.
|
||||||
|
|
||||||
|
| Variable | Default | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `TERDUT_ADDR` | `:8080` | TCP address to listen on |
|
||||||
|
| `TERDUT_DB_DSN` | — | **Required.** Postgres connection string, e.g. `postgres://terdut:secret@localhost:5432/terdut?sslmode=require` |
|
||||||
|
| `TERDUT_ARCHIVE_AFTER` | `168h` (7d) | **seed.** How long a resolved alert or incident stays in the default list before being auto-archived |
|
||||||
|
| `TERDUT_STALE_AFTER` | `6h` | **seed.** How long a firing alert may go without a refreshing webhook before it is treated as resolved — **must exceed your Alertmanager `repeat_interval`** |
|
||||||
|
| `TERDUT_NTFY_URL` | — | ntfy server to publish push notifications to. Empty disables notifications entirely |
|
||||||
|
| `TERDUT_NTFY_TOKEN` | — | Bearer token for an access-controlled ntfy |
|
||||||
|
| `TERDUT_NTFY_FALLBACK_TOPIC` | — | Topic used when nobody is on call |
|
||||||
|
| `TERDUT_PUBLIC_URL` | — | Base URL a phone uses to reach this server: the notification's link into the web UI, its Acknowledge button, and whether the session cookie is `Secure` |
|
||||||
|
| `TERDUT_NOTIFY_REPEAT` | `15m` | **seed.** How long an incident may sit unacknowledged before it is paged again. `0` notifies once and never repeats |
|
||||||
|
| `TERDUT_PASSWORD_LOGIN` | `true` | `false` refuses password login and password sign-up (`403`), leaving single sign-on the only way in. Refused at startup unless SSO is configured |
|
||||||
|
| `TERDUT_TRUSTED_PROXIES` | `1` | How many reverse proxies in front of the server append to `X-Forwarded-For`; the per-address rate limits use the entry that many hops from the right. `0` ignores the header |
|
||||||
|
| `TERDUT_OPERATOR_KEY` | — | At least 32 characters. When set, the instance-scoped service account `terdut-operator` is created if missing and its `seed` key replaced with this value at every start — how terdut-operator authenticates without a bootstrap handshake. An instance-scoped account acts as owner of every team (team configuration) but is not a member of any, so it reads no incidents |
|
||||||
|
| `TERDUT_OPERATOR_MODE` | `false` | Declares this install gitops-managed: a session's or a user's own API key's writes to teams, escalation policies, dead man's switches and integrations are refused (`403 reason:"operator_managed"`); a [service account](./api.md#service-accounts)'s are not. Team membership and the schedule stay editable regardless |
|
||||||
|
| `TERDUT_OIDC_ISSUER` | — | Turns single sign-on on. The provider's issuer URL; discovery is read from `<issuer>/.well-known/openid-configuration`. See [Single sign-on](./single-sign-on.md#single-sign-on-oidc) |
|
||||||
|
| `TERDUT_OIDC_CLIENT_ID` / `TERDUT_OIDC_CLIENT_SECRET` | — | **Required with an issuer.** The confidential client registered at the provider. Keep the secret in a Secret, not in values |
|
||||||
|
| `TERDUT_OIDC_NAME` | `SSO` | What the sign-in button calls the provider |
|
||||||
|
| `TERDUT_OIDC_SCOPES` | `openid profile email` | Scopes requested, comma or space separated. Authentik puts `groups` behind `profile` |
|
||||||
|
| `TERDUT_OIDC_USERNAME_CLAIM` / `_EMAIL_CLAIM` / `_GROUPS_CLAIM` | `preferred_username` / `email` / `groups` | ID token claims read for the username, email and groups |
|
||||||
|
| `TERDUT_OIDC_TRUST_EMAIL` | `false` | Link a first sign-in to an existing local user by email even if the provider does not mark the address verified |
|
||||||
|
| `TERDUT_OIDC_ALLOWED_GROUPS` | — | Comma-separated. Only people in one of these may sign in. Empty admits everybody the provider authenticates |
|
||||||
|
| `TERDUT_OIDC_ADMIN_GROUP` | — | Members are system administrators |
|
||||||
|
| `TERDUT_OIDC_SESSION_MAX_AGE` | `12h` | Hard ceiling on a session made by an SSO sign-in |
|
||||||
|
|
||||||
|
Durations use Go syntax (`30m`, `12h`, `168h`). An unparseable value falls back to the default.
|
||||||
|
|
||||||
|
Note that `TERDUT_STALE_AFTER` and a dead man's switch timeout point in opposite directions. Staleness
|
||||||
|
is a generous grace period around a `repeat_interval` you do not control; a dead man's switch is a
|
||||||
|
deadline you set deliberately, and the heartbeat's route is configured to beat faster than it.
|
||||||
|
|
||||||
|
In the Helm chart the two sweeper durations are set via `sweeper.staleAfter` and `sweeper.archiveAfter`, notifications via the `notify.*` values, single sign-on via `oidc.*` and `passwordLogin`, and operator mode via `operatorMode`.
|
||||||
@@ -0,0 +1,81 @@
|
|||||||
|
# Dead man's switches
|
||||||
|
|
||||||
|
_Detecting that alerts have stopped arriving._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Everything above assumes alerts arrive. If Prometheus stops evaluating, or
|
||||||
|
Alertmanager cannot reach this server, nothing arrives — and silence looks
|
||||||
|
exactly like everything being fine. A dead man's switch inverts the handling for
|
||||||
|
one designated alert so that silence is the signal:
|
||||||
|
|
||||||
|
- **receiving** it opens no incident, and
|
||||||
|
- the **absence** of it does.
|
||||||
|
|
||||||
|
kube-prometheus-stack already ships the alert for this. `Watchdog` is
|
||||||
|
`expr: vector(1)`, so it fires permanently and is re-sent forever; it is worth
|
||||||
|
nothing unless something downstream notices it stop. It is the usual first switch.
|
||||||
|
|
||||||
|
**Switches belong to a team**, which decides which of its own alerts are
|
||||||
|
heartbeats and how long a silence has to last. Each **switch** is a row of its
|
||||||
|
own — a name, one matcher, a timeout and a severity — so switches in one team
|
||||||
|
can have different deadlines. An owner adds and removes them on **Team →
|
||||||
|
Switches**, which lists each with a status (**healthy**, **dead**, or
|
||||||
|
**dormant** until its first heartbeat), when it was last heard from, and when it
|
||||||
|
last opened an incident; a matcher that several clusters satisfy is broken down
|
||||||
|
per cluster. The API is `POST`/`DELETE /api/teams/{teamID}/deadman/switches`. A
|
||||||
|
missed heartbeat opens an incident in the team whose integration received it.
|
||||||
|
Removing a switch stops the watching; an incident it already opened stays open
|
||||||
|
until somebody resolves it.
|
||||||
|
|
||||||
|
A new team watches nothing until its owner (or terdut-operator, from a
|
||||||
|
`TerdutTeam`) adds a switch: inheriting an install-wide heartbeat would page a
|
||||||
|
new team about a source it has never heard of.
|
||||||
|
|
||||||
|
A matcher is a set of exact label conditions, one of which must be the
|
||||||
|
`alertname`, , one matcher per switch, `,` between the label conditions:
|
||||||
|
|
||||||
|
```
|
||||||
|
alertname=Watchdog,cluster=prod
|
||||||
|
```
|
||||||
|
|
||||||
|
**The unit of monitoring is the fingerprint, not the alert name.** Two clusters
|
||||||
|
sending the same `Watchdog` are two independent switches, so a healthy one can
|
||||||
|
never mask a dead one.
|
||||||
|
|
||||||
|
## The lifecycle
|
||||||
|
|
||||||
|
A switch is **dormant** until its first heartbeat arrives. A configured matcher
|
||||||
|
that has never been heard from opens nothing, so a fresh deploy or a restored
|
||||||
|
database does not page. It also means a matcher that never matches anything is
|
||||||
|
silently inert.
|
||||||
|
|
||||||
|
Once armed, the sweeper declares it **dead** when either the heartbeat has not
|
||||||
|
been refreshed within the switch's `timeout_seconds`, or Alertmanager explicitly
|
||||||
|
resolved it — the sender saying the heartbeat stopped needs no further waiting.
|
||||||
|
That opens an incident at the switch's `severity`, assigned and paged like any
|
||||||
|
other, and marks the heartbeat alert `"resolution_source": "deadman"` so the
|
||||||
|
alert list stops claiming a dead switch is firing.
|
||||||
|
|
||||||
|
It **recovers** when the heartbeat starts arriving again: the incident resolves
|
||||||
|
with `"resolution_source": "recovered"` and the all-clear goes to whoever was
|
||||||
|
paged.
|
||||||
|
|
||||||
|
Resolving the incident by hand sticks, the same way it does for an alert-backed
|
||||||
|
one. While the switch stays silent nothing new opens — so a decommissioned
|
||||||
|
source is a one-time page rather than a nag. The switch **re-arms** on the next
|
||||||
|
heartbeat: come back and die again, and that is a new incident.
|
||||||
|
|
||||||
|
## Two things to know
|
||||||
|
|
||||||
|
A switch's timeout must be **shorter** than the `repeat_interval` of the
|
||||||
|
route carrying the heartbeat, which is the exact opposite of
|
||||||
|
`TERDUT_STALE_AFTER`. Inheriting a default `repeat_interval` of 4h gives you a
|
||||||
|
switch that takes four hours to notice anything, so give the heartbeat
|
||||||
|
[its own route](./alertmanager.md#alertmanager-configuration). Matched alerts are exempt from
|
||||||
|
stale-alert expiry — a heartbeat answers to its own timeout and nothing else.
|
||||||
|
|
||||||
|
A dead man's switch incident has **no member alerts**:
|
||||||
|
`GET /api/incidents/{id}/alerts` returns an empty list. There is no alert
|
||||||
|
describing the problem, because the problem is that no alert arrived. What
|
||||||
|
happened is on the timeline instead, as a `deadman_silent` event carrying the age
|
||||||
|
of the last heartbeat, and the heartbeat's labels are on the incident's
|
||||||
|
`group_labels`.
|
||||||
@@ -0,0 +1,74 @@
|
|||||||
|
# Deployment
|
||||||
|
|
||||||
|
_Running the server in a container and on Kubernetes with the Helm chart._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
## Docker
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker build -t terdut-server .
|
||||||
|
docker run -p 8080:8080 \
|
||||||
|
-e TERDUT_DB_DSN='postgres://terdut:secret@host.docker.internal:5432/terdut?sslmode=disable' \
|
||||||
|
terdut-server
|
||||||
|
```
|
||||||
|
|
||||||
|
The server creates its own schema on startup and needs a reachable Postgres; it stores nothing on
|
||||||
|
disk, so there is no volume to mount.
|
||||||
|
|
||||||
|
## Kubernetes
|
||||||
|
|
||||||
|
A Helm chart is published from this repository as an OCI artifact, versioned in lockstep
|
||||||
|
with the app — chart `x.y.z` is always app `vx.y.z`:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
helm upgrade --install terdut-server oci://git.ryuvia.com/niklas/terdut-server \
|
||||||
|
--version 0.9.2 \
|
||||||
|
--namespace terdut-server --create-namespace \
|
||||||
|
--set networking.hostname=terdut.example.com
|
||||||
|
```
|
||||||
|
|
||||||
|
The chart expects a [Gateway API](https://gateway-api.sigs.k8s.io/) Gateway named `envoy-main` in
|
||||||
|
the `envoy-gateway-system` namespace to already exist — it renders an `HTTPRoute` against it rather
|
||||||
|
than an `Ingress`. TLS is terminated at the gateway, so the server itself never sees a certificate.
|
||||||
|
|
||||||
|
| Value | Default | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `networking.hostname` | `terdut.example.com` | Hostname the `HTTPRoute` serves |
|
||||||
|
| `networking.listener` | `""` | Gateway listener (`sectionName`) to bind to. Empty attaches to every matching listener, **including plaintext HTTP** — set it to the HTTPS listener's name to serve TLS only |
|
||||||
|
| `networking.servicePort` | `8080` | Port the route forwards to; keep in sync with `service.port` |
|
||||||
|
| `bootstrap.enabled` | `true` | Runs a post-install hook that creates the first user and stores its API key in the `<release>-admin-key` Secret. Already-bootstrapped servers are left alone |
|
||||||
|
| `database.dsn` | `""` | **Required.** Postgres DSN, with no password in it. The chart provisions no database |
|
||||||
|
| `database.passwordSecret.name` | `""` | Secret supplying `PGPASSWORD`. With the Zalando postgres operator, the Secret it generates for the role |
|
||||||
|
| `database.passwordSecret.key` | `password` | Key within that Secret |
|
||||||
|
|
||||||
|
The API key travels in an `Authorization: Bearer` header, so set `networking.listener` whenever the
|
||||||
|
hostname is reachable outside a trusted network.
|
||||||
|
|
||||||
|
### The database
|
||||||
|
|
||||||
|
The chart provisions no database: it takes a DSN and expects a Postgres that already exists. In this
|
||||||
|
cluster the wrapper chart declares an `acid.zalan.do/v1 postgresql` CR; anywhere else, any reachable
|
||||||
|
Postgres 14+ will do.
|
||||||
|
|
||||||
|
The DSN carries no password. pgx falls back to libpq's environment variables for whatever the DSN
|
||||||
|
leaves out, so the password arrives as `PGPASSWORD` from a Secret and never appears in values, in
|
||||||
|
the rendered manifest or in `kubectl describe pod`. With the postgres operator that Secret is the
|
||||||
|
one it generates for the role, so a rebuild mints a new password with nothing to keep in sync —
|
||||||
|
the same wiring miniflux uses.
|
||||||
|
|
||||||
|
The server migrates its own schema on startup, so a new database only has to exist and be writable.
|
||||||
|
|
||||||
|
### Backups
|
||||||
|
|
||||||
|
Postgres is backed up where it runs, not from here. The database pod carries a
|
||||||
|
[k8up](https://k8up.io/) `k8up.io/backupcommand` annotation that streams a `pg_dump`, the same way
|
||||||
|
gitea and immich do in this cluster.
|
||||||
|
|
||||||
|
## On Kubernetes with the operator
|
||||||
|
|
||||||
|
[terdut-operator](https://git.ryuvia.com/niklas/terdut-operator) runs a server for you from a
|
||||||
|
`TerdutServer` object and manages its teams, escalation ladders, dead man's switches and alert
|
||||||
|
sources as Kubernetes objects. It hands the server a generated key through `TERDUT_OPERATOR_KEY`
|
||||||
|
(see [Configuration](./configuration.md) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)), and
|
||||||
|
the server then treats configuration as operator-managed (`TERDUT_OPERATOR_MODE`), refusing edits
|
||||||
|
made by hand in the web UI. Use the Helm chart above for a plain install, the operator when you
|
||||||
|
want that configuration in gitops.
|
||||||
@@ -0,0 +1,93 @@
|
|||||||
|
# Development and releasing
|
||||||
|
|
||||||
|
_Building, testing and releasing the server._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
## Upgrading
|
||||||
|
|
||||||
|
The schema is a single baseline (`internal/db/migrations/001_schema.sql`) and no
|
||||||
|
release has shipped yet, so there is no upgrade path from earlier development
|
||||||
|
databases: start from an empty one. Changes after the first release arrive as
|
||||||
|
new numbered migrations.
|
||||||
|
|
||||||
|
## Development
|
||||||
|
|
||||||
|
```bash
|
||||||
|
make test-db # start a local Postgres for the tests (podman or docker)
|
||||||
|
make test # run all tests
|
||||||
|
go build ./... # compile all packages
|
||||||
|
go run ./cmd/terdut # run locally (needs TERDUT_DB_DSN)
|
||||||
|
```
|
||||||
|
|
||||||
|
The tests need a real Postgres, because the server does — there is no in-memory Postgres.
|
||||||
|
`TERDUT_TEST_DSN` says where it is, `make test-db` starts
|
||||||
|
one on port 5433 and prints the DSN, and `make test-db-stop` removes it. Each test gets its
|
||||||
|
own schema on that server, so tests cannot see each other's rows. An unset `TERDUT_TEST_DSN`
|
||||||
|
fails the suite rather than skipping it: a run that quietly tests nothing is worse than one
|
||||||
|
that does not run.
|
||||||
|
|
||||||
|
`make fmt lint test helm-lint` is the gate. It mirrors `.gitea/workflows/ci.yaml` step for
|
||||||
|
step, so a green run here means a green pipeline — with one deliberate exception: `make test`
|
||||||
|
adds `-race`, which CI does not. The sweeper, the notifier goroutine and the dead man's switch
|
||||||
|
sweep all run concurrently against the same database, and a race between them would surface as
|
||||||
|
a flaky incident in production rather than as a red build.
|
||||||
|
|
||||||
|
The web UI lives in `internal/web/static/` as plain HTML, CSS and ES modules,
|
||||||
|
embedded into the binary with `go:embed`. It has no build step and no npm, so
|
||||||
|
editing a file and restarting the server is the whole loop.
|
||||||
|
|
||||||
|
## Releasing
|
||||||
|
|
||||||
|
```
|
||||||
|
push or PR → ci.yaml gofmt, go vet, go test -race
|
||||||
|
govulncheck, gitleaks
|
||||||
|
helm lint + render
|
||||||
|
push tag vX.Y.Z → release.yaml the same gate, then publish:
|
||||||
|
git.ryuvia.com/niklas/terdut-server:vX.Y.Z
|
||||||
|
oci://git.ryuvia.com/niklas/terdut-server X.Y.Z
|
||||||
|
then trivy-scan the pushed image
|
||||||
|
PR to Ryuvia/charts → bump the wrapper chart to X.Y.Z; on merge
|
||||||
|
Flux reconciles and the release rolls out
|
||||||
|
```
|
||||||
|
|
||||||
|
Both artifacts go to the **personal** Gitea namespace rather than `ryuvia`, because Gitea
|
||||||
|
scopes package visibility to the owner with no per-package override — so `ryuvia/*` is private
|
||||||
|
because the org is. Publishing to `niklas` keeps them anonymously pullable, which is why no
|
||||||
|
pull secret is needed in the cluster. Same reasoning, and the same choice, as riksdata and
|
||||||
|
rd-web.
|
||||||
|
|
||||||
|
Saying **"Release"** runs all three rows: the `release` skill commits, pushes, tags, waits for
|
||||||
|
the pipeline, and opens the `Ryuvia/charts` PR, stopping before the merge. See
|
||||||
|
`~/.claude/skills/release/`, or `.release.conf` here for this repo's part of it.
|
||||||
|
|
||||||
|
The chart is published **only** from the tag, by the `chart` job. There used to be a second
|
||||||
|
publisher on every `charts/**` push to main, and the two raced for the same chart version with
|
||||||
|
different answers — chart 0.9.0 went out reading `appVersion: "latest"` that way. One
|
||||||
|
publisher, triggered by the tag (`766f439`). The cost is that a chart-only change has no
|
||||||
|
version of its own and rides the next app tag.
|
||||||
|
|
||||||
|
Both workflows are thin drivers over the Makefile: `ci.yaml` runs `make fmt lint test` and
|
||||||
|
`make helm-lint`, `release.yaml` adds `make binaries`, `make push`, `make helm-package` and
|
||||||
|
`make helm-push`. That is deliberate — it is what makes a green local gate and a green
|
||||||
|
pipeline the same code rather than two descriptions of it, and it is how riksdata and rd-web
|
||||||
|
have always worked.
|
||||||
|
|
||||||
|
`make push` builds and pushes in one step, unlike those two, because the image is
|
||||||
|
`linux/amd64,linux/arm64` and buildx cannot load a multi-platform result into the local image
|
||||||
|
store. `make build` stays single-platform and local-only. Both refuse `VERSION=dev`:
|
||||||
|
publishing is one command, so it is also one command to run by accident. Publishing happens
|
||||||
|
by pushing a tag.
|
||||||
|
|
||||||
|
Two things the release process needs to know about this repo:
|
||||||
|
|
||||||
|
- **The image scan runs after publishing**, like riksdata's and rd-web's: trivy cannot read
|
||||||
|
a locally built image on this runner, so it pulls the pushed one. A red `scan-image` means
|
||||||
|
do not bump the wrapper chart to that version — it cannot unpublish anything. The image is
|
||||||
|
`FROM scratch`, so trivy sees exactly one target, the Go binary and its module graph.
|
||||||
|
- **The wrapper chart's `values.yaml` has two `tag:` lines** — the app image and the python
|
||||||
|
backup sidecar — so `chart-bump` is given `--image` to say which one moves. Once the wrapper
|
||||||
|
chart drops the sidecar and declares a `postgresql` CR instead, there is one `tag:` line
|
||||||
|
again, and `--image` becomes belt and braces.
|
||||||
|
|
||||||
|
The wrapper chart must have **its own `version:` bumped in the same commit**. Flux reconciles
|
||||||
|
with `reconcileStrategy: ChartVersion`, so a chart whose version did not change produces no
|
||||||
|
new artifact and the change is never deployed — with no error anywhere.
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
# Escalation
|
||||||
|
|
||||||
|
_Escalation ladders: who is paged next when nobody acknowledges._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Without a ladder, an unacknowledged incident re-pages the same topic every
|
||||||
|
`notify_repeat` forever. That is a louder version of the same silence: if the
|
||||||
|
person on call is asleep, out of signal, or has left, nothing else happens.
|
||||||
|
|
||||||
|
A team can configure an ordered ladder instead. Each level has a timeout and a
|
||||||
|
set of targets, and a target is either a named person or **whoever the team's
|
||||||
|
rota says is on call today** — the target that keeps working when the rota
|
||||||
|
changes and nobody remembers to edit the policy.
|
||||||
|
|
||||||
|
```
|
||||||
|
level 1 5m oncall the rota gets first refusal
|
||||||
|
level 2 5m user:bob then a named second
|
||||||
|
then repeat_count more rounds
|
||||||
|
then the team's fallback topic, once
|
||||||
|
```
|
||||||
|
|
||||||
|
When a level's timeout passes with the incident still `triggered`, the next
|
||||||
|
level is paged. Off the end of the ladder the whole thing runs again
|
||||||
|
`repeat_count` times, and after that the team's `fallback_topic` is paged once
|
||||||
|
as the end of the line. The incident stays open throughout: running out of
|
||||||
|
people to wake is not the same as somebody answering.
|
||||||
|
|
||||||
|
**Acknowledging or resolving stops it**, which is the point — continuing to wake
|
||||||
|
people after somebody has said "I have this" is how a tool teaches people to
|
||||||
|
mute it. **Snoozing pauses it**: a deliberate "not now" holds the ladder where
|
||||||
|
it is, and it resumes when the snooze runs out.
|
||||||
|
|
||||||
|
Every step is on the incident's timeline with the level and the names it woke,
|
||||||
|
so somebody reading it afterwards can tell why their phone rang at 04:00. A
|
||||||
|
level whose targets are all unreachable — no ntfy topic, a disabled account, an
|
||||||
|
empty rota — is recorded as `nobody reachable` and the ladder moves on rather
|
||||||
|
than stalling on a rung that cannot ring.
|
||||||
|
|
||||||
|
**Reminders and escalation never both run.** A team with a ladder gets
|
||||||
|
escalation; a team without keeps the reminder behaviour exactly as it was. Two
|
||||||
|
pages for one silence is the surest way to get a tool muted.
|
||||||
|
|
||||||
|
The ladder's `fallback_topic` is per team, unlike `TERDUT_NTFY_FALLBACK_TOPIC`,
|
||||||
|
which is the install-wide topic used when an incident opens with nobody on call.
|
||||||
|
They answer different questions: one is "nobody was scheduled", the other is
|
||||||
|
"everybody scheduled has been tried".
|
||||||
|
After Width: | Height: | Size: 23 KiB |
|
After Width: | Height: | Size: 55 KiB |
|
After Width: | Height: | Size: 88 KiB |
|
After Width: | Height: | Size: 52 KiB |
|
After Width: | Height: | Size: 53 KiB |
|
After Width: | Height: | Size: 100 KiB |
|
After Width: | Height: | Size: 93 KiB |
|
After Width: | Height: | Size: 94 KiB |
|
After Width: | Height: | Size: 42 KiB |
|
After Width: | Height: | Size: 35 KiB |
|
After Width: | Height: | Size: 44 KiB |
@@ -0,0 +1,113 @@
|
|||||||
|
# Alerts and incidents
|
||||||
|
|
||||||
|
_How alerts become incidents and how incidents are worked, notified, escalated and expired._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
There are two objects, and the difference between them is the whole design.
|
||||||
|
|
||||||
|
**An alert is Alertmanager's record.** It has two states, `firing` and
|
||||||
|
`resolved`, one row per fingerprint, and no human ever writes to it. The API
|
||||||
|
exposes alerts read-only.
|
||||||
|
|
||||||
|
**An incident is the work item.** It goes `triggered → acknowledged → resolved`,
|
||||||
|
carries an assignee, a snooze, notes and a timeline, and is the only thing people
|
||||||
|
act on. Many alerts belong to one incident.
|
||||||
|
|
||||||
|
## Correlation uses Alertmanager's `groupKey`
|
||||||
|
|
||||||
|
Alertmanager has already grouped alerts according to the `group_by` routing tree
|
||||||
|
you configured, and it sends the resulting `groupKey` and `groupLabels` on every
|
||||||
|
webhook. Incidents adopt that answer rather than re-grouping alerts a second
|
||||||
|
time — if you want different correlation, change `group_by` in
|
||||||
|
`alertmanager.yml` and terdut follows.
|
||||||
|
|
||||||
|
At most one incident is open per `groupKey` at a time. Alerts firing in a group
|
||||||
|
that already has an open incident join it. The incident's `severity` is a
|
||||||
|
high-water mark — the highest `severity` label any of its alerts has carried — so
|
||||||
|
an incident that hit `critical` still reads as critical after the critical alert
|
||||||
|
clears.
|
||||||
|
|
||||||
|
## Several clusters, one team
|
||||||
|
|
||||||
|
A team with one Alertmanager per Kubernetes cluster, each posting to its own
|
||||||
|
source, needs two settings or the clusters run together.
|
||||||
|
|
||||||
|
1. Give every alert a `cluster` label at the source. In Prometheus that is
|
||||||
|
`externalLabels: {cluster: prod-eu}` (kube-prometheus-stack:
|
||||||
|
`prometheus.prometheusSpec.externalLabels`).
|
||||||
|
2. Add `cluster` to `group_by` in `alertmanager.yml`.
|
||||||
|
|
||||||
|
The second one is the one that matters. Incidents are matched on the team and
|
||||||
|
Alertmanager's `groupKey`, and the `groupKey` does not include external labels:
|
||||||
|
without `cluster` in `group_by`, the same alert in two clusters has the same
|
||||||
|
key and joins one incident. With it, each cluster gets its own, `cluster` is in
|
||||||
|
the incident's `group_labels`, and the web UI shows it as a coloured chip on the
|
||||||
|
queue, the incident and the alert list, instead of leaving it in the title.
|
||||||
|
An alert that is not grouped by `cluster` still shows the chip on the alert
|
||||||
|
list, which reads the label from the alert itself.
|
||||||
|
|
||||||
|
The queue has a cluster dropdown once there are two or more values to choose
|
||||||
|
between. It filters on the incident's `cluster` group label
|
||||||
|
(`GET /api/incidents?cluster=...`), so it only sees incidents grouped by it.
|
||||||
|
|
||||||
|
## An incident opens only on a new occurrence
|
||||||
|
|
||||||
|
An incident opens when an alert **transitions into firing**: a fingerprint that
|
||||||
|
was never seen, an alert with a newer `startsAt`, or a resolved alert that
|
||||||
|
started again. The unchanged firing notifications Alertmanager re-sends every
|
||||||
|
`repeat_interval` are none of those, and open nothing.
|
||||||
|
|
||||||
|
This is what makes closing an incident by hand mean something. Without the rule,
|
||||||
|
`POST /api/incidents/{id}/resolve` would be undone by the next re-send of an
|
||||||
|
alert that never stopped firing.
|
||||||
|
|
||||||
|
## Leaving the open state
|
||||||
|
|
||||||
|
- **Automatically**, once every alert under the incident has stopped firing —
|
||||||
|
whether by a resolved webhook or by the sweeper's
|
||||||
|
[stale-alert expiry](#stale-alert-expiry). The incident gets
|
||||||
|
`"resolution_source": "alerts"`.
|
||||||
|
- **By hand**, via `POST /api/incidents/{id}/resolve`
|
||||||
|
(`"resolution_source": "manual"`). This is **terminal**: a later occurrence in
|
||||||
|
that group opens a *new* incident rather than reopening this one. If the alert
|
||||||
|
underneath never stops firing, the incident stays closed — that is what
|
||||||
|
resolving by hand asserts.
|
||||||
|
- **On recovery**, for a [dead man's switch](./dead-mans-switch.md) incident whose
|
||||||
|
heartbeat started arriving again (`"resolution_source": "recovered"`). These
|
||||||
|
incidents have no member alerts, so the automatic cascade above cannot reach
|
||||||
|
them.
|
||||||
|
|
||||||
|
To quieten an incident you expect to come back, snooze it instead
|
||||||
|
(`POST /api/incidents/{id}/snooze`). A snooze hides the incident from the default
|
||||||
|
list without closing it, and expires by simply falling into the past.
|
||||||
|
|
||||||
|
## On-call assignment
|
||||||
|
|
||||||
|
A new incident is assigned to whoever holds today's schedule entry at the moment
|
||||||
|
it opens (`GET /api/schedule/current`). If nobody is scheduled it opens
|
||||||
|
unassigned. Reassign with `POST /api/incidents/{id}/assign`.
|
||||||
|
|
||||||
|
One person holds a given day, so `POST /api/schedule` refuses a date somebody
|
||||||
|
already has: taking a shift off the person expecting to be paged for it should
|
||||||
|
not be something a plain call does by accident. Pass `"replace": true` to take
|
||||||
|
them anyway. Either way the whole request is one transaction — a week where some
|
||||||
|
days are free and some are taken moves as a unit, and a failure leaves the rota
|
||||||
|
exactly as it was rather than with a hole in it.
|
||||||
|
|
||||||
|
## Stale alert expiry
|
||||||
|
|
||||||
|
A resolved webhook is the only signal that an alert has stopped firing, so a
|
||||||
|
notification that is dropped, silenced, or lost to a restart would otherwise pin
|
||||||
|
that alert as firing forever. A background sweeper resolves firing alerts that
|
||||||
|
Alertmanager has stopped refreshing, using either signal:
|
||||||
|
|
||||||
|
- the `endsAt` watermark on the last notification has passed, or
|
||||||
|
- no webhook has refreshed the alert within `TERDUT_STALE_AFTER`.
|
||||||
|
|
||||||
|
Alertmanager re-sends firing notifications every `repeat_interval`, which is what
|
||||||
|
keeps a live alert fresh — so `TERDUT_STALE_AFTER` must be comfortably larger
|
||||||
|
than your `repeat_interval` (default 4h), or live alerts will be resolved
|
||||||
|
prematurely. Alerts resolved this way are marked `"resolution_source": "expiry"`
|
||||||
|
to distinguish them from a real Alertmanager resolve (`"alertmanager"`).
|
||||||
|
|
||||||
|
An expiry cascades: once it leaves an incident with nothing firing under it, the
|
||||||
|
incident resolves too, in the same sweep.
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
# Push notifications
|
||||||
|
|
||||||
|
_Pages through ntfy, who gets them and how to acknowledge from the notification._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
With `TERDUT_NTFY_URL` set, an incident that opens is pushed to the on-call
|
||||||
|
person's phone through [ntfy](https://ntfy.sh). Everybody sets their own topic
|
||||||
|
under *Account* in the web UI, where a **Send a test push** button proves it
|
||||||
|
before an incident has to; `PUT /api/users/{id}/notify` is the same thing over
|
||||||
|
the API, and an administrator may set somebody else's. A user with no topic
|
||||||
|
falls back to `TERDUT_NTFY_FALLBACK_TOPIC`, as does an incident that opens with
|
||||||
|
nobody on call. If neither yields a topic, nothing is queued.
|
||||||
|
|
||||||
|
The **server** is the install's one ntfy, from `TERDUT_NTFY_URL`, and is not
|
||||||
|
something a user picks. Only the topic is per-person.
|
||||||
|
|
||||||
|
A topic is a shared secret with the ntfy server: anyone who knows it can both
|
||||||
|
read the pages and publish to it, so an unguessable one is worth the trouble.
|
||||||
|
That is also why the topic never appears in an incident's timeline, which every
|
||||||
|
API key can read.
|
||||||
|
|
||||||
|
Three things get pushed:
|
||||||
|
|
||||||
|
- **triggered** — an incident opened. Priority follows severity (`critical` maps
|
||||||
|
to ntfy's max priority, the one that overrides the phone's quiet settings).
|
||||||
|
- **reminder** — the incident is still `triggered` after `TERDUT_NOTIFY_REPEAT`.
|
||||||
|
Repeats until somebody acts. Acknowledging, snoozing, resolving or archiving
|
||||||
|
all stop it — snooze is the mute button.
|
||||||
|
- **resolved** — every alert under the incident stopped firing. Only sent to
|
||||||
|
whoever was paged in the first place, and only for the automatic cascade:
|
||||||
|
resolving by hand pushes nothing, since the person who did it already knows.
|
||||||
|
|
||||||
|
Notifications carry an **Acknowledge** button that acknowledges the incident
|
||||||
|
without opening anything. It POSTs to `/api/notify/ack/{token}`, an
|
||||||
|
unauthenticated route authorised by the 256-bit token in its path — minted fresh
|
||||||
|
per notification, scoped to one incident and one action, and valid for 24 hours.
|
||||||
|
A real API key is never put in a notification, because the message is stored on
|
||||||
|
the ntfy server and cached on the device.
|
||||||
|
|
||||||
|
The token is **not** consumed by use. Acknowledging is idempotent, so a token
|
||||||
|
stays valid for its full 24 hours and a second tap is a no-op that reports the
|
||||||
|
incident's current state rather than an error — which is what you want when a
|
||||||
|
tap is retried on a flaky mobile connection. What bounds it is scope, not a use
|
||||||
|
count: one incident, one action, one day. Expired tokens are purged by the
|
||||||
|
sweeper.
|
||||||
|
|
||||||
|
Two consequences worth planning for:
|
||||||
|
|
||||||
|
- `/api/notify/ack/{token}` **must stay publicly reachable**, or the button will
|
||||||
|
not work when the responder is off your network.
|
||||||
|
- Notifications sent to the fallback topic carry **no** Acknowledge button. The
|
||||||
|
topic is shared, and a button on it would let any subscriber acknowledge as
|
||||||
|
somebody else.
|
||||||
|
|
||||||
|
Delivery is a queue, not an inline call: the webhook writes a row and a
|
||||||
|
background notifier sends it within 30 seconds, retrying with exponential
|
||||||
|
backoff up to 8 attempts. Nothing about ingestion blocks on ntfy being reachable.
|
||||||
|
|
||||||
|
Every delivery is recorded on the incident's timeline: a `notified` event once
|
||||||
|
ntfy accepts the publish, and a `notify_failed` event when a notification
|
||||||
|
exhausts its retries. Written from the result rather than at enqueue, so the
|
||||||
|
timeline says what actually happened — and a page that never landed is visible
|
||||||
|
instead of looking the same as one that did.
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
# Single sign-on (OIDC)
|
||||||
|
|
||||||
|
_Signing in through an OpenID Connect provider, and mapping its groups to teams and administrators._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
terdut can sign people in through any OpenID Connect provider; the examples use
|
||||||
|
[Authentik](https://goauthentik.io/). Groups at the provider decide who may sign
|
||||||
|
in, which teams they belong to and whether they administer the install, much as
|
||||||
|
Grafana's OAuth role and org mapping does. Password login keeps working alongside
|
||||||
|
it unless you turn it off.
|
||||||
|
|
||||||
|
**At the provider**, create an OAuth2/OpenID provider and an application for it:
|
||||||
|
a *confidential* client, redirect URI `<TERDUT_PUBLIC_URL>/api/oidc/callback`, and
|
||||||
|
the `openid`, `profile` and `email` scopes. The issuer is the application's, e.g.
|
||||||
|
`https://auth.example.com/application/o/terdut/`. Then set:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
TERDUT_PUBLIC_URL=https://terdut.example.com
|
||||||
|
TERDUT_OIDC_ISSUER=https://auth.example.com/application/o/terdut/
|
||||||
|
TERDUT_OIDC_CLIENT_ID=terdut
|
||||||
|
TERDUT_OIDC_CLIENT_SECRET=...
|
||||||
|
TERDUT_OIDC_ALLOWED_GROUPS=terdut-users,terdut-admins
|
||||||
|
TERDUT_OIDC_ADMIN_GROUP=terdut-admins
|
||||||
|
```
|
||||||
|
|
||||||
|
Which team a group grants is not server-wide config: each team names its own
|
||||||
|
group(s), set by that team's own owner (or an administrator) from its Members
|
||||||
|
tab, or `PUT /api/teams/{teamID}/oidc-groups {"member_group":"sre","owner_group":"sre-leads"}`.
|
||||||
|
A team must already exist before a group can grant access to it — the sync
|
||||||
|
never creates one.
|
||||||
|
|
||||||
|
The web UI's sign-in page shows a "Sign in with <name>" button (a plain link to
|
||||||
|
`/api/oidc/login`) above the password form, or instead of it when
|
||||||
|
`TERDUT_PASSWORD_LOGIN=false`; it asks `GET /api/auth/config` what the server offers
|
||||||
|
(`password_login`, `oidc.enabled`, `oidc.name`). A refused sign-in comes back to that
|
||||||
|
page with the reason spelled out. Access the groups grant is badged **SSO** on the
|
||||||
|
Team, Admin and per-user pages, with its edit and remove controls disabled, and the
|
||||||
|
Account page does not offer to set a password nobody could use.
|
||||||
|
|
||||||
|
**What a sign-in does**
|
||||||
|
|
||||||
|
1. *Who.* The provider's `(issuer, subject)` is the identity. The first time, a
|
||||||
|
user is found by email — only when the provider marks it verified, or
|
||||||
|
`TERDUT_OIDC_TRUST_EMAIL` is set — or created with no password. A username taken
|
||||||
|
by somebody else gets a numeric suffix (`alice-2`). Username and email follow the
|
||||||
|
provider at each sign-in. Authentik reports `email_verified` as false unless
|
||||||
|
configured otherwise, so linking existing users usually needs
|
||||||
|
`TERDUT_OIDC_TRUST_EMAIL=true`.
|
||||||
|
2. *Whether.* With `TERDUT_OIDC_ALLOWED_GROUPS` set, somebody in none of them is
|
||||||
|
refused and nothing is created.
|
||||||
|
3. *What.* The administrator flag follows `TERDUT_OIDC_ADMIN_GROUP`. Team roles
|
||||||
|
follow each team's own `oidc_member_group`/`oidc_owner_group`; where both of a
|
||||||
|
team's groups match, the owner group wins.
|
||||||
|
|
||||||
|
**Managed access.** What the sync grants is marked as managed by single sign-on,
|
||||||
|
and only that is ever changed by it. It is added at sign-in, and removed at the
|
||||||
|
next sign-in after the group is gone, even if that leaves a team without an owner
|
||||||
|
(an administrator can always repair a team) — the provider is the source of truth
|
||||||
|
for what it grants, so the last-owner and last-administrator guards do not apply.
|
||||||
|
Memberships and administrators added by hand are left alone; the exception is a
|
||||||
|
hand-added member whose team's own group grants a *higher* role, who is raised and
|
||||||
|
from then on managed. Editing managed access by hand (`POST` or `DELETE` on a
|
||||||
|
team's members, revoking an SSO-granted administrator) is refused with `409`, since
|
||||||
|
the next sign-in would undo it.
|
||||||
|
|
||||||
|
> **Upgrading past migration 013: reconfigure every team's groups.**
|
||||||
|
> `TERDUT_OIDC_GROUP_MAPPINGS` is gone, and the sync no longer creates a team by
|
||||||
|
> name. Group-to-team-role mapping is now each team's own setting — an owner sets
|
||||||
|
> it from the Members tab, or `PUT /api/teams/{teamID}/oidc-groups`. Until a team's
|
||||||
|
> owner does that, an OIDC-sourced membership in it is dropped at that user's next
|
||||||
|
> SSO sign-in, the same as any other loss of group access. Set every team's groups
|
||||||
|
> before affected users next sign in, to avoid a visible gap in access.
|
||||||
|
|
||||||
|
**How fast changes arrive.** Groups are read only at sign-in. A session made by an
|
||||||
|
SSO sign-in has a hard ceiling (`TERDUT_OIDC_SESSION_MAX_AGE`, default 12h) that
|
||||||
|
sliding never extends, so a change at the provider reaches terdut within that time.
|
||||||
|
Password sessions are unaffected.
|
||||||
|
|
||||||
|
> **API keys are not revoked when somebody is removed at the provider.** terdut
|
||||||
|
> holds no refresh token and never asks the provider again, so a person removed
|
||||||
|
> from every allowed group loses their sessions within `TERDUT_OIDC_SESSION_MAX_AGE`
|
||||||
|
> and cannot sign in again, but keeps any API key they made (the TUI and scripts use
|
||||||
|
> them) until an administrator disables the user in terdut.
|
||||||
|
|
||||||
|
**Signing in from a terminal.** A client with no browser of its own, such as the
|
||||||
|
TUI over SSH, signs in with a device code, run by terdut itself so the terminal
|
||||||
|
never talks to the provider:
|
||||||
|
|
||||||
|
1. The terminal calls `POST /api/oidc/device` and shows the person a link
|
||||||
|
(`<TERDUT_PUBLIC_URL>/device?code=XXXX-XXXX`) and the code.
|
||||||
|
2. On any device the person opens the link, signs in (by the provider or by
|
||||||
|
password, whatever the login page offers), sees the code and the account, and
|
||||||
|
presses **Approve**. Only a browser session can approve; an API key cannot.
|
||||||
|
3. The terminal polls `POST /api/oidc/device/token` every 5 seconds and is given the
|
||||||
|
ordinary `terdut_session` cookie once. A person who signs in through the provider
|
||||||
|
gets the same `TERDUT_OIDC_SESSION_MAX_AGE` ceiling on the terminal's session as
|
||||||
|
on their browser's.
|
||||||
|
|
||||||
|
A login expires after 10 minutes. `GET /api/auth/config` reports `device_login`.
|
||||||
|
|
||||||
|
**If the provider is down**, terdut still starts (discovery is fetched on first
|
||||||
|
use) and password login is the way in. With `TERDUT_PASSWORD_LOGIN=false` that way
|
||||||
|
is closed: set it back to `true`. The first administrator comes from the bootstrap
|
||||||
|
endpoint, and stays a manual administrator that no group can revoke; on an SSO-only
|
||||||
|
install set `bootstrap.enabled: false` in the chart if you don't want that account,
|
||||||
|
or keep it and never give it a password.
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
# The web UI
|
||||||
|
|
||||||
|
_What the web UI offers, how sign-in and sessions work, and the Team and Admin tabs._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
The server serves a web UI at `/`: the incident queue, each incident's alerts
|
||||||
|
and timeline with every action (acknowledge, assign, snooze, note, resolve,
|
||||||
|
archive), who is on call, the alert feed, and an *Account* tab for your own
|
||||||
|
password and the ntfy topic your pages go to. It is built for a phone first. On a phone
|
||||||
|
it navigates through a hamburger menu and has a sticky action bar, it follows the
|
||||||
|
system's dark mode, and it can be added to the home screen. From 900px wide it switches
|
||||||
|
to a sidebar with the queue and the incident side by side. The Stats page shows
|
||||||
|
incident counts, MTTA and MTTR, and alert frequency by name, hour and day over a
|
||||||
|
chosen range.
|
||||||
|
|
||||||
|
You sign in with a username and password. Users have no password until one is
|
||||||
|
set, and a user without one can only use API keys:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# an admin sets someone's first password with their API key
|
||||||
|
curl -X PUT http://localhost:8080/api/users/2/password \
|
||||||
|
-H "Authorization: Bearer $KEY" -H "Content-Type: application/json" \
|
||||||
|
-d '{"password": "<at least 10 characters>"}'
|
||||||
|
```
|
||||||
|
|
||||||
|
After that, users change it themselves under *Account*. Changing your own
|
||||||
|
password requires the current one.
|
||||||
|
|
||||||
|
How a browser stays signed in:
|
||||||
|
|
||||||
|
- A successful login sets an `HttpOnly`, `SameSite=Lax` session cookie. It lasts
|
||||||
|
30 days and slides forward while it is used, so an on-call phone stays signed
|
||||||
|
in.
|
||||||
|
- The cookie is marked `Secure` when `TERDUT_PUBLIC_URL` starts with `https://`,
|
||||||
|
so set it to the HTTPS address. TLS terminates at the gateway and the server
|
||||||
|
itself only ever sees plain HTTP.
|
||||||
|
- Requests authenticated by the cookie are checked for cross-origin use (Go's
|
||||||
|
`http.CrossOriginProtection`). That is the CSRF guard. Bearer-key clients are
|
||||||
|
not affected.
|
||||||
|
- Setting a password signs that user out everywhere else.
|
||||||
|
- Ten failed logins for one username within 15 minutes lock that username for
|
||||||
|
the rest of the window.
|
||||||
|
|
||||||
|
With `TERDUT_PUBLIC_URL` set, tapping a push notification opens the incident in
|
||||||
|
the web UI (`/incidents/{id}`).
|
||||||
|
|
||||||
|
A **Team** tab holds everything a team owns, in five sub-sections with a URL
|
||||||
|
each and a strip across the top to move between them: the on-call rota
|
||||||
|
(`/team/rota`), the membership (`/team/members`), the escalation ladder
|
||||||
|
(`/team/escalation`), the alert sources with their keys (`/team/sources`) and
|
||||||
|
the dead man's switches (`/team/deadman`). `/team` itself is an overview — who
|
||||||
|
is on call today, how many members and owners, how many ladder levels, how many
|
||||||
|
keys and how many switches — so a page fetches only what it shows. An owner
|
||||||
|
edits it; a member sees the same pages read-only, because the server refuses
|
||||||
|
their writes anyway. Somebody in more than one team picks between them above
|
||||||
|
the strip, since the choice changes the subject of all five.
|
||||||
|
|
||||||
|
The rota is a month at a time, one coloured initial per day with a legend
|
||||||
|
underneath, and it says how many days are left uncovered — the question a rota
|
||||||
|
is read for is who holds which stretch, and a run of one colour answers it
|
||||||
|
where a list of dates does not. An owner taps a day to hand it to somebody or
|
||||||
|
empty it, and fills a whole shift from the range form folded in below.
|
||||||
|
|
||||||
|
The **Admin** tab appears only for a system administrator, and holds what
|
||||||
|
belongs to the whole server rather than to one team. It has three sub-sections,
|
||||||
|
each with a URL of its own and a strip across the top to move between them:
|
||||||
|
every team (`/admin/teams`), every user (`/admin/users`), and the settings that
|
||||||
|
used to be environment variables (`/admin/settings`). `/admin` itself is an
|
||||||
|
overview — how many of each, and what each section is for. Adding somebody is
|
||||||
|
minting them an invite link into a team, rather than creating a bare account:
|
||||||
|
the person who accepts it picks their own password, so one never passes through
|
||||||
|
an administrator, and the link carries the team, so they land somewhere with a
|
||||||
|
queue in it. That happens on the team's own page, since an invite is a fact
|
||||||
|
about a team; the user list points there rather than asking which team beside a
|
||||||
|
form.
|
||||||
|
|
||||||
|
A name in the team list opens **that team's page**, at `/admin/teams/{id}`: when it
|
||||||
|
was created, how many are in it and how much is open, a field to rename it, the
|
||||||
|
members with their roles, the invites into it, and deletion. The member list is the
|
||||||
|
one thing there that needed a new endpoint — `GET /api/teams/{id}/members` is
|
||||||
|
member-only and answers `404` to an administrator who is not in the team, which is
|
||||||
|
the rule and not an oversight, so the page reads `GET /api/admin/teams/{id}` instead.
|
||||||
|
An administrator still sees none of that team's incidents, alerts or rota.
|
||||||
|
|
||||||
|
A name in the user list opens **that person's page**, at `/admin/users/{id}`: their
|
||||||
|
email and when they joined, where their notifications go, whether they are an
|
||||||
|
administrator, whether the account is disabled, the teams they are in with their
|
||||||
|
role in each, a password field for a first or forgotten one, and deletion. It is
|
||||||
|
the one place membership is edited from the person's side — the Team tab answers
|
||||||
|
"who is in this team", and answering "which teams is this person in" there means
|
||||||
|
visiting each team in turn.
|
||||||
@@ -87,7 +87,7 @@ func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc
|
|||||||
respond(w, http.StatusUnauthorized, errResp("unknown integration key"))
|
respond(w, http.StatusUnauthorized, errResp("unknown integration key"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
receiveWebhook(w, r, db, notify, src)
|
receiveWebhook(w, r, db, notify, src)
|
||||||
|
|||||||
@@ -90,7 +90,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc {
|
|||||||
fmt.Sprintf("%s WHERE %s ORDER BY a.received_at DESC LIMIT %s", alertSelectFrom, clause, args.add(limit)),
|
fmt.Sprintf("%s WHERE %s ORDER BY a.received_at DESC LIMIT %s", alertSelectFrom, clause, args.add(limit)),
|
||||||
args.all()...)
|
args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -99,7 +99,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc {
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
a, err := scanAlert(rows)
|
a, err := scanAlert(rows)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
alerts = append(alerts, a)
|
alerts = append(alerts, a)
|
||||||
@@ -121,7 +121,7 @@ func handleGetAlert(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, a)
|
respond(w, http.StatusOK, a)
|
||||||
|
|||||||
@@ -78,6 +78,10 @@ func newTSWith(t *testing.T, deadman api.DeadmanConfig, cfg api.NotifyConfig, co
|
|||||||
|
|
||||||
s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman}
|
s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman}
|
||||||
|
|
||||||
|
// A fresh install has no team, so the tests that want "the" team make it
|
||||||
|
// here: it is id 1, owned by the admin, which is what defaultTeam names.
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Default"}), &struct{}{})
|
||||||
|
|
||||||
var integration struct {
|
var integration struct {
|
||||||
Key string `json:"key"`
|
Key string `json:"key"`
|
||||||
}
|
}
|
||||||
@@ -768,7 +772,7 @@ func TestWebhook_IgnoresOutOfOrderRetry(t *testing.T) {
|
|||||||
// An expiry resolve writes ends_at as an upper bound, not an observed end: an
|
// An expiry resolve writes ends_at as an upper bound, not an observed end: an
|
||||||
// Alertmanager watermark already on the row is preserved, and a row that never
|
// Alertmanager watermark already on the row is preserved, and a row that never
|
||||||
// carried one is stamped at sweep time. Clients are told to read it that way —
|
// carried one is stamped at sweep time. Clients are told to read it that way —
|
||||||
// see "resolution_source says how much to trust ends_at" in the README.
|
// see "resolution_source says how much to trust ends_at" in docs/api.md.
|
||||||
func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
|
func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
|
|
||||||
@@ -813,7 +817,7 @@ func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
|
|||||||
//
|
//
|
||||||
// received_at is documented as a public liveness signal, so these lock the
|
// received_at is documented as a public liveness signal, so these lock the
|
||||||
// behaviour clients are told they may rely on. See "received_at is a liveness
|
// behaviour clients are told they may rely on. See "received_at is a liveness
|
||||||
// heartbeat" in the README and the comment on models.Alert.ReceivedAt.
|
// heartbeat" in docs/api.md and the comment on models.Alert.ReceivedAt.
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
// The heartbeat itself: an unchanged firing notification — what Alertmanager
|
// The heartbeat itself: an unchanged firing notification — what Alertmanager
|
||||||
@@ -875,3 +879,38 @@ func TestStats_ByDayReturnsSevenSlots(t *testing.T) {
|
|||||||
t.Errorf("expected 7 day slots, got %d", len(slots))
|
t.Errorf("expected 7 day slots, got %d", len(slots))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Two simultaneous bootstraps on an empty install must not both win.
|
||||||
|
func TestBootstrap_ConcurrentCallsCreateOneAdmin(t *testing.T) {
|
||||||
|
database := newTestDB(t)
|
||||||
|
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig(), "test"))
|
||||||
|
t.Cleanup(srv.Close)
|
||||||
|
|
||||||
|
const n = 8
|
||||||
|
codes := make(chan int, n)
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
go func(i int) {
|
||||||
|
body, _ := json.Marshal(map[string]string{"username": fmt.Sprintf("u%d", i), "email": fmt.Sprintf("u%d@x.com", i)})
|
||||||
|
resp, err := http.Post(srv.URL+"/api/bootstrap", "application/json", bytes.NewReader(body))
|
||||||
|
if err != nil {
|
||||||
|
codes <- 0
|
||||||
|
return
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
codes <- resp.StatusCode
|
||||||
|
}(i)
|
||||||
|
}
|
||||||
|
created := 0
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
if <-codes == http.StatusCreated {
|
||||||
|
created++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var users int
|
||||||
|
if err := database.QueryRow("SELECT COUNT(*) FROM users").Scan(&users); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if created != 1 || users != 1 {
|
||||||
|
t.Errorf("expected exactly one bootstrap to win, got %d created and %d users", created, users)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -154,9 +154,7 @@ func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip
|
|||||||
}
|
}
|
||||||
|
|
||||||
// staleAlertIDs reads the ids in one go and closes the cursor before the caller
|
// staleAlertIDs reads the ids in one go and closes the cursor before the caller
|
||||||
// writes. Under SQLite's single connection an open read would have blocked the
|
// writes, which keeps the write off a cursor the same transaction is walking.
|
||||||
// update outright; with a pool it is no longer a deadlock, but reading the set
|
|
||||||
// first still keeps the write off a cursor the same transaction is walking.
|
|
||||||
func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) {
|
func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) {
|
||||||
rows, err := db.QueryContext(ctx, `
|
rows, err := db.QueryContext(ctx, `
|
||||||
SELECT id FROM alerts
|
SELECT id FROM alerts
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ import (
|
|||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
|
"sync/atomic"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
@@ -28,6 +29,9 @@ const (
|
|||||||
// sessionTouchEvery bounds how often a request may slide the expiry.
|
// sessionTouchEvery bounds how often a request may slide the expiry.
|
||||||
sessionTouchEvery = time.Hour
|
sessionTouchEvery = time.Hour
|
||||||
|
|
||||||
|
// keyTouchEvery is the same bound for an API key's last_used_at.
|
||||||
|
keyTouchEvery = 5 * time.Minute
|
||||||
|
|
||||||
minPasswordLen = 10
|
minPasswordLen = 10
|
||||||
// maxPasswordLen is bcrypt's limit; it rejects longer input outright.
|
// maxPasswordLen is bcrypt's limit; it rejects longer input outright.
|
||||||
maxPasswordLen = 72
|
maxPasswordLen = 72
|
||||||
@@ -126,14 +130,32 @@ func purgeRateLimits(ctx context.Context, db *sql.DB) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// trustedProxies is how many X-Forwarded-For hops clientAddr trusts. Set once
|
||||||
|
// by NewRouter from config.
|
||||||
|
var trustedProxies atomic.Int64
|
||||||
|
|
||||||
// clientAddr is the address a login is counted against. Behind the gateway
|
// clientAddr is the address a login is counted against. Behind the gateway
|
||||||
// RemoteAddr is the gateway itself, so the first X-Forwarded-For hop is used
|
// RemoteAddr is the gateway itself, so the client address is read from
|
||||||
// when present. It can be forged, but only to dodge the address limit; the
|
// X-Forwarded-For, counting trustedProxies entries from the right: each trusted
|
||||||
// per-username limit does not depend on it.
|
// proxy appends the address it saw, so the entries to the left of those are
|
||||||
|
// client-supplied and could be forged to dodge the limit.
|
||||||
func clientAddr(r *http.Request) string {
|
func clientAddr(r *http.Request) string {
|
||||||
if xff := r.Header.Get("X-Forwarded-For"); xff != "" {
|
if n := int(trustedProxies.Load()); n > 0 {
|
||||||
first, _, _ := strings.Cut(xff, ",")
|
var hops []string
|
||||||
return strings.TrimSpace(first)
|
for _, v := range r.Header.Values("X-Forwarded-For") {
|
||||||
|
for _, h := range strings.Split(v, ",") {
|
||||||
|
if h = strings.TrimSpace(h); h != "" {
|
||||||
|
hops = append(hops, h)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(hops) > 0 {
|
||||||
|
i := len(hops) - n
|
||||||
|
if i < 0 {
|
||||||
|
i = 0
|
||||||
|
}
|
||||||
|
return hops[i]
|
||||||
|
}
|
||||||
}
|
}
|
||||||
host, _, err := net.SplitHostPort(r.RemoteAddr)
|
host, _, err := net.SplitHostPort(r.RemoteAddr)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -241,7 +263,7 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
|
|||||||
"SELECT id, password_hash FROM users WHERE username = $1", username,
|
"SELECT id, password_hash FROM users WHERE username = $1", username,
|
||||||
).Scan(&userID, &hash)
|
).Scan(&userID, &hash)
|
||||||
if err != nil && !errors.Is(err, sql.ErrNoRows) {
|
if err != nil && !errors.Is(err, sql.ErrNoRows) {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -258,13 +280,13 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
|
|||||||
limiter.clear(r.Context(), userKey)
|
limiter.clear(r.Context(), userKey)
|
||||||
|
|
||||||
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, userID)
|
user, err := fetchUser(r.Context(), db, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, meResponse{User: user, HasPassword: true})
|
respond(w, http.StatusOK, meResponse{User: user, HasPassword: true})
|
||||||
@@ -320,7 +342,7 @@ func handleMe(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
user, err := fetchUser(r.Context(), db, caller.ID)
|
user, err := fetchUser(r.Context(), db, caller.ID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
var hash sql.NullString
|
var hash sql.NullString
|
||||||
@@ -332,7 +354,7 @@ func handleMe(db *sql.DB) http.HandlerFunc {
|
|||||||
// transient database problem, not a missing user — worth a 500
|
// transient database problem, not a missing user — worth a 500
|
||||||
// rather than silently answering "no password, not dismissed",
|
// rather than silently answering "no password, not dismissed",
|
||||||
// which a client would otherwise take at face value.
|
// which a client would otherwise take at face value.
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, meResponse{
|
respond(w, http.StatusOK, meResponse{
|
||||||
@@ -384,7 +406,7 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -397,30 +419,30 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
hash, err := hashPassword(req.Password)
|
hash, err := hashPassword(req.Password)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback()
|
defer tx.Rollback()
|
||||||
|
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"UPDATE users SET password_hash = $1 WHERE id = $2", hash, id); err != nil {
|
"UPDATE users SET password_hash = $1 WHERE id = $2", hash, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
keep, _ := sessionFromContext(r.Context()) // zero when changed with an API key
|
keep, _ := sessionFromContext(r.Context()) // zero when changed with an API key
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"DELETE FROM sessions WHERE user_id = $1 AND id != $2", id, keep); err != nil {
|
"DELETE FROM sessions WHERE user_id = $1 AND id != $2", id, keep); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
|||||||
@@ -2,7 +2,6 @@ package api
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"fmt"
|
|
||||||
|
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
)
|
)
|
||||||
@@ -61,7 +60,7 @@ func (c Caller) IsAdmin() bool {
|
|||||||
// by being a human (and becomes its owner as a side effect), an
|
// by being a human (and becomes its owner as a side effect), an
|
||||||
// instance-scoped service account creates one with no human owner at all;
|
// instance-scoped service account creates one with no human owner at all;
|
||||||
// the two paths are not interchangeable, so this predicate must not also
|
// the two paths are not interchangeable, so this predicate must not also
|
||||||
// admit a human admin the way MayActAsInstanceAdmin deliberately does.
|
// admit a human admin the way an administrator check would.
|
||||||
func (c Caller) IsInstanceServiceAccount() bool {
|
func (c Caller) IsInstanceServiceAccount() bool {
|
||||||
return c.sa != nil && c.sa.scope == models.ServiceAccountScopeInstance
|
return c.sa != nil && c.sa.scope == models.ServiceAccountScopeInstance
|
||||||
}
|
}
|
||||||
@@ -110,24 +109,6 @@ func (c Caller) ServiceAccountName() (string, bool) {
|
|||||||
return c.sa.name, true
|
return c.sa.name, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// Identity is a stable, log/audit-facing string distinguishing a human
|
|
||||||
// caller from a service account — "user:42" or "service-account:7". Not
|
|
||||||
// wired into any database column — incidents.go's acknowledged_by/
|
|
||||||
// incident_events.user_id use AsHuman()/ServiceAccountID() directly against
|
|
||||||
// the parallel *_service_account_id columns (migration 015) instead, since a
|
|
||||||
// column needs the id, not this rendered string. assigned_to stays
|
|
||||||
// human-only and out of scope (terdut-server#25's follow-up).
|
|
||||||
func (c Caller) Identity() string {
|
|
||||||
switch {
|
|
||||||
case c.user != nil:
|
|
||||||
return fmt.Sprintf("user:%d", c.user.ID)
|
|
||||||
case c.sa != nil:
|
|
||||||
return fmt.Sprintf("service-account:%d", c.sa.id)
|
|
||||||
default:
|
|
||||||
return "unknown"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func callerFromContext(ctx context.Context) (Caller, bool) {
|
func callerFromContext(ctx context.Context) (Caller, bool) {
|
||||||
c, ok := ctx.Value(ctxCaller).(Caller)
|
c, ok := ctx.Value(ctxCaller).(Caller)
|
||||||
return c, ok
|
return c, ok
|
||||||
|
|||||||
@@ -65,28 +65,6 @@ func (m DeadmanMatcher) matches(labels map[string]string) bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// DeadmanConfig is the server-wide default a team's switches are seeded from:
|
|
||||||
// the environment's matchers, timeout and severity. Switches themselves are rows
|
|
||||||
// of a team's own — see DeadmanSwitch — and this is only how a fresh install
|
|
||||||
// starts out.
|
|
||||||
type DeadmanConfig struct {
|
|
||||||
Matchers []DeadmanMatcher
|
|
||||||
|
|
||||||
// Timeout is how long a matched alert may go without a refreshing webhook
|
|
||||||
// before it is declared dead. It must be shorter than Alertmanager's
|
|
||||||
// repeat_interval for the heartbeat's route, which is what refreshes it.
|
|
||||||
// Zero disables dead man's switch handling entirely.
|
|
||||||
Timeout time.Duration
|
|
||||||
|
|
||||||
// Severity is the severity every dead man's switch incident opens at. These
|
|
||||||
// incidents have no member alerts to derive one from, and the heartbeat's
|
|
||||||
// own severity label is meaningless — Watchdog ships as "none".
|
|
||||||
Severity string
|
|
||||||
}
|
|
||||||
|
|
||||||
// enabled reports whether there is anything to watch.
|
|
||||||
func (c DeadmanConfig) enabled() bool { return c.Timeout > 0 && len(c.Matchers) > 0 }
|
|
||||||
|
|
||||||
// DeadmanSwitch inverts the handling of the alerts it matches: receiving one
|
// DeadmanSwitch inverts the handling of the alerts it matches: receiving one
|
||||||
// opens nothing, and the absence of one opens an incident.
|
// opens nothing, and the absence of one opens an incident.
|
||||||
//
|
//
|
||||||
@@ -160,46 +138,6 @@ func parseDeadmanMatcher(entry string) (DeadmanMatcher, error) {
|
|||||||
return m, nil
|
return m, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// ParseDeadmanConfig reads the matcher list from its configured form:
|
|
||||||
// ";" separates matchers, and each is parsed as parseDeadmanMatcher does.
|
|
||||||
//
|
|
||||||
// A malformed or alertname-less entry is dropped rather than fatal, following
|
|
||||||
// config.duration's rule that one bad tuning knob should not take the server
|
|
||||||
// down. Silence would be worse here than elsewhere, though — a typo that
|
|
||||||
// disarms the switch is exactly the failure this feature exists to catch — so
|
|
||||||
// the matchers that survived are logged.
|
|
||||||
func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig {
|
|
||||||
cfg := DeadmanConfig{Timeout: timeout, Severity: severity}
|
|
||||||
|
|
||||||
for _, entry := range strings.Split(matchers, ";") {
|
|
||||||
entry = strings.TrimSpace(entry)
|
|
||||||
if entry == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
m, err := parseDeadmanMatcher(entry)
|
|
||||||
if err != nil {
|
|
||||||
log.Printf("deadman: ignoring matcher %q: %v", entry, err)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
cfg.Matchers = append(cfg.Matchers, m)
|
|
||||||
}
|
|
||||||
|
|
||||||
switch {
|
|
||||||
case timeout <= 0:
|
|
||||||
log.Print("deadman: disabled (timeout is zero)")
|
|
||||||
case len(cfg.Matchers) == 0:
|
|
||||||
log.Print("deadman: disabled (no usable matchers)")
|
|
||||||
default:
|
|
||||||
rendered := make([]string, 0, len(cfg.Matchers))
|
|
||||||
for _, m := range cfg.Matchers {
|
|
||||||
rendered = append(rendered, m.String())
|
|
||||||
}
|
|
||||||
log.Printf("deadman: default for new teams: %s, timeout %s, severity %s",
|
|
||||||
strings.Join(rendered, "; "), timeout, severity)
|
|
||||||
}
|
|
||||||
return cfg
|
|
||||||
}
|
|
||||||
|
|
||||||
// deadmanAlert is one heartbeat: the alert row carrying its last sighting, and
|
// deadmanAlert is one heartbeat: the alert row carrying its last sighting, and
|
||||||
// the switch that claimed it.
|
// the switch that claimed it.
|
||||||
type deadmanAlert struct {
|
type deadmanAlert struct {
|
||||||
@@ -490,56 +428,6 @@ func deadmanSets(ctx context.Context, db *sql.DB) (map[int64]deadmanSet, error)
|
|||||||
return scanDeadmanSwitches(rows)
|
return scanDeadmanSwitches(rows)
|
||||||
}
|
}
|
||||||
|
|
||||||
// deadmanSeededKey is the settings row that records the environment defaults
|
|
||||||
// were handed out. Without it, a team that deleted its last switch would get
|
|
||||||
// the default back on the next restart.
|
|
||||||
const deadmanSeededKey = "deadman_seeded"
|
|
||||||
|
|
||||||
// SeedDeadmanConfigs gives every team the server's environment defaults as
|
|
||||||
// switches, exactly once per install, so a fresh install watches Watchdog
|
|
||||||
// without anybody setting it up.
|
|
||||||
//
|
|
||||||
// Once seeded it never runs again: a team's switches are its own, and a redeploy
|
|
||||||
// must not quietly put the environment's value back over an owner's edit or
|
|
||||||
// deletion. Installs that upgraded from per-team configuration were already
|
|
||||||
// seeded, which migration 009 records.
|
|
||||||
//
|
|
||||||
// A team created after that gets none and watches nothing until its owner says
|
|
||||||
// otherwise. That is deliberate: inheriting an install-wide heartbeat would page
|
|
||||||
// a new team about a source it has never heard of, and a switch nobody chose is
|
|
||||||
// the kind that gets muted rather than fixed.
|
|
||||||
func SeedDeadmanConfigs(ctx context.Context, db *sql.DB, cfg DeadmanConfig) error {
|
|
||||||
if !cfg.enabled() {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
tx, err := db.BeginTx(ctx, nil)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
defer tx.Rollback() //nolint:errcheck
|
|
||||||
|
|
||||||
res, err := tx.ExecContext(ctx,
|
|
||||||
"INSERT INTO settings (key, value) VALUES ($1, '1') ON CONFLICT (key) DO NOTHING",
|
|
||||||
deadmanSeededKey)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, m := range cfg.Matchers {
|
|
||||||
if _, err := tx.ExecContext(ctx, `
|
|
||||||
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
|
||||||
SELECT id, $1, $1, $2, $3 FROM teams`,
|
|
||||||
m.config(), int64(cfg.Timeout.Seconds()), cfg.Severity); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return tx.Commit()
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Status
|
// Status
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
package api_test
|
package api_test
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
|
||||||
"net/http"
|
"net/http"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
@@ -122,8 +121,11 @@ func TestDeadman_MixedGroupExcludesHeartbeat(t *testing.T) {
|
|||||||
t.Fatalf("expected 1 incident for the real alert, got %d", got)
|
t.Fatalf("expected 1 incident for the real alert, got %d", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
var alerts []map[string]any
|
var incident struct {
|
||||||
decode(t, s.req(t, http.MethodGet, "/api/incidents/1/alerts", nil), &alerts)
|
Alerts []map[string]any `json:"alerts"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/incidents/1", nil), &incident)
|
||||||
|
alerts := incident.Alerts
|
||||||
if len(alerts) != 1 {
|
if len(alerts) != 1 {
|
||||||
t.Fatalf("expected 1 member alert, got %d", len(alerts))
|
t.Fatalf("expected 1 member alert, got %d", len(alerts))
|
||||||
}
|
}
|
||||||
@@ -717,26 +719,3 @@ func TestDeadman_DeleteIsScopedToTheTeam(t *testing.T) {
|
|||||||
t.Errorf("expected no switches, got %d", got)
|
t.Errorf("expected no switches, got %d", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// The environment's defaults are handed out once and then belong to the teams.
|
|
||||||
func TestDeadman_SeedRunsOnce(t *testing.T) {
|
|
||||||
s := newTS(t)
|
|
||||||
cfg := api.ParseDeadmanConfig("alertname=Watchdog", time.Hour, "critical")
|
|
||||||
|
|
||||||
if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil {
|
|
||||||
t.Fatalf("seed: %v", err)
|
|
||||||
}
|
|
||||||
if got := len(listSwitches(t, s)); got != 1 {
|
|
||||||
t.Fatalf("the first seed should add the default, got %d switches", got)
|
|
||||||
}
|
|
||||||
|
|
||||||
// The owner deletes it; a restart must not put it back.
|
|
||||||
id := int64(listSwitches(t, s)[0]["id"].(float64))
|
|
||||||
s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(id), nil).Body.Close()
|
|
||||||
if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil {
|
|
||||||
t.Fatalf("seed again: %v", err)
|
|
||||||
}
|
|
||||||
if got := len(listSwitches(t, s)); got != 0 {
|
|
||||||
t.Errorf("a second seed resurrected %d switch(es)", got)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -81,7 +81,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http
|
|||||||
|
|
||||||
deviceCode, deviceHash, err := randomToken()
|
deviceCode, deviceHash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -105,7 +105,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http
|
|||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
log.Printf("device login: start: %v", err)
|
log.Printf("device login: start: %v", err)
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -160,7 +160,7 @@ func handleDeviceDecision(db *sql.DB, approve bool) http.HandlerFunc {
|
|||||||
WHERE user_code = $3 AND status = 'pending' AND expires_at > $4`,
|
WHERE user_code = $3 AND status = 'pending' AND expires_at > $4`,
|
||||||
status, caller.ID, code, time.Now().Unix())
|
status, caller.ID, code, time.Now().Unix())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -190,7 +190,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
|
|||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
@@ -206,7 +206,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -226,11 +226,11 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
|
|||||||
}
|
}
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"UPDATE device_logins SET last_polled_at = $1 WHERE device_hash = $2", now.Unix(), hash); err != nil {
|
"UPDATE device_logins SET last_polled_at = $1 WHERE device_hash = $2", now.Unix(), hash); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusAccepted, map[string]string{"status": "pending"})
|
respond(w, http.StatusAccepted, map[string]string{"status": "pending"})
|
||||||
@@ -240,7 +240,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
|
|||||||
// Approved. Single use: the row goes before the session is made, so two
|
// Approved. Single use: the row goes before the session is made, so two
|
||||||
// racing polls cannot both be given one.
|
// racing polls cannot both be given one.
|
||||||
if _, err := tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash); err != nil {
|
if _, err := tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
var disabled, sso bool
|
var disabled, sso bool
|
||||||
@@ -252,7 +252,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if disabled {
|
if disabled {
|
||||||
@@ -269,12 +269,12 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
|
|||||||
}
|
}
|
||||||
if err := startSessionCapped(w, r, db, userID.Int64, publicURL, maxAge); err != nil {
|
if err := startSessionCapped(w, r, db, userID.Int64, publicURL, maxAge); err != nil {
|
||||||
log.Printf("device login: start session: %v", err)
|
log.Printf("device login: start session: %v", err)
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, err := fetchUser(r.Context(), db, userID.Int64)
|
user, err := fetchUser(r.Context(), db, userID.Int64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, meResponse{User: user, HasPassword: false})
|
respond(w, http.StatusOK, meResponse{User: user, HasPassword: false})
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package api
|
|||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
|
"errors"
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
@@ -343,12 +344,12 @@ func handleGetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
view, err := escalationStatus(r.Context(), db, teamID, policy)
|
view, err := escalationStatus(r.Context(), db, teamID, policy)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, view)
|
respond(w, http.StatusOK, view)
|
||||||
@@ -543,6 +544,10 @@ type escalationLevelJSON struct {
|
|||||||
type escalationTargetJSON struct {
|
type escalationTargetJSON struct {
|
||||||
Kind string `json:"kind"`
|
Kind string `json:"kind"`
|
||||||
UserID *int64 `json:"user_id,omitempty"`
|
UserID *int64 `json:"user_id,omitempty"`
|
||||||
|
// Username is accepted in place of user_id on a PUT, and resolved to the
|
||||||
|
// id before anything is stored. It is never returned: the stored form is
|
||||||
|
// the id, which survives a rename.
|
||||||
|
Username string `json:"username,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type escalationJSON struct {
|
type escalationJSON struct {
|
||||||
@@ -600,6 +605,27 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
for i, l := range req.Levels {
|
for i, l := range req.Levels {
|
||||||
|
for j, t := range l.Targets {
|
||||||
|
if t.Username == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if t.Kind != "user" || t.UserID != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("username belongs on a user target, instead of user_id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var id int64
|
||||||
|
err := db.QueryRowContext(r.Context(), "SELECT id FROM users WHERE username = $1", t.Username).Scan(&id)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("unknown user "+strconv.Quote(t.Username)))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Levels[i].Targets[j].UserID = &id
|
||||||
|
req.Levels[i].Targets[j].Username = ""
|
||||||
|
}
|
||||||
if l.TimeoutSeconds <= 0 {
|
if l.TimeoutSeconds <= 0 {
|
||||||
respond(w, http.StatusBadRequest, errResp("every level needs a timeout"))
|
respond(w, http.StatusBadRequest, errResp("every level needs a timeout"))
|
||||||
return
|
return
|
||||||
@@ -632,7 +658,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
@@ -645,13 +671,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
fallback_topic = excluded.fallback_topic,
|
fallback_topic = excluded.fallback_topic,
|
||||||
updated_at = excluded.updated_at`,
|
updated_at = excluded.updated_at`,
|
||||||
teamID, req.RepeatCount, req.FallbackTopic); err != nil {
|
teamID, req.RepeatCount, req.FallbackTopic); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// The levels are replaced, not merged; the cascade takes the targets.
|
// The levels are replaced, not merged; the cascade takes the targets.
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil {
|
"DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -661,7 +687,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
INSERT INTO escalation_levels (team_id, position, timeout_seconds)
|
INSERT INTO escalation_levels (team_id, position, timeout_seconds)
|
||||||
VALUES ($1, $2, $3) RETURNING id`,
|
VALUES ($1, $2, $3) RETURNING id`,
|
||||||
teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil {
|
teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
for _, t := range l.Targets {
|
for _, t := range l.Targets {
|
||||||
@@ -676,13 +702,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, escalationResponse(policy, teamID))
|
respond(w, http.StatusOK, escalationResponse(policy, teamID))
|
||||||
|
|||||||
@@ -502,3 +502,45 @@ func TestEscalation_StatusWithoutALadder(t *testing.T) {
|
|||||||
t.Errorf("a team with no ladder should read as empty, got %+v", v)
|
t.Errorf("a team with no ladder should read as empty, got %+v", v)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A user target may name the person instead of carrying an id; the server
|
||||||
|
// resolves it and stores the id.
|
||||||
|
func TestEscalation_UserTargetByUsername(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
id := teamUser(t, s, "alice", "")
|
||||||
|
|
||||||
|
put := func(username string) *http.Response {
|
||||||
|
return s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", map[string]any{
|
||||||
|
"repeat_count": 0,
|
||||||
|
"levels": []map[string]any{{
|
||||||
|
"timeout_seconds": 300,
|
||||||
|
"targets": []map[string]any{{"kind": "user", "username": username}},
|
||||||
|
}},
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
resp := put("alice")
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode >= 300 {
|
||||||
|
t.Fatalf("PUT by username: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var got struct {
|
||||||
|
Levels []struct {
|
||||||
|
Targets []struct {
|
||||||
|
UserID *int64 `json:"user_id"`
|
||||||
|
Username string `json:"username"`
|
||||||
|
} `json:"targets"`
|
||||||
|
} `json:"levels"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil), &got)
|
||||||
|
if len(got.Levels) != 1 || len(got.Levels[0].Targets) != 1 ||
|
||||||
|
got.Levels[0].Targets[0].UserID == nil || *got.Levels[0].Targets[0].UserID != id {
|
||||||
|
t.Errorf("expected the target stored as user %d, got %+v", id, got)
|
||||||
|
}
|
||||||
|
|
||||||
|
bad := put("nobody")
|
||||||
|
bad.Body.Close()
|
||||||
|
if bad.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("unknown username should be a 400, got %d", bad.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// DeadmanConfig is a test fixture only: a set of matchers with one timeout and
|
||||||
|
// severity, turned into switches over the API by the test helpers. Production
|
||||||
|
// has no server-wide default any more -- switches belong to teams.
|
||||||
|
type DeadmanConfig struct {
|
||||||
|
Matchers []DeadmanMatcher
|
||||||
|
Timeout time.Duration
|
||||||
|
Severity string
|
||||||
|
}
|
||||||
|
|
||||||
|
// ParseDeadmanConfig reads a ";"-separated matcher list the way the removed
|
||||||
|
// environment variable did, dropping malformed entries.
|
||||||
|
func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig {
|
||||||
|
cfg := DeadmanConfig{Timeout: timeout, Severity: severity}
|
||||||
|
for _, entry := range strings.Split(matchers, ";") {
|
||||||
|
entry = strings.TrimSpace(entry)
|
||||||
|
if entry == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if m, err := parseDeadmanMatcher(entry); err == nil {
|
||||||
|
cfg.Matchers = append(cfg.Matchers, m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return cfg
|
||||||
|
}
|
||||||
@@ -5,6 +5,7 @@ import (
|
|||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"errors"
|
"errors"
|
||||||
|
"github.com/go-chi/chi/v5"
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
@@ -17,8 +18,7 @@ import (
|
|||||||
// sqlArgs accumulates query arguments and hands back the placeholder for each.
|
// sqlArgs accumulates query arguments and hands back the placeholder for each.
|
||||||
//
|
//
|
||||||
// Postgres numbers its placeholders, so a dynamically assembled WHERE clause has
|
// Postgres numbers its placeholders, so a dynamically assembled WHERE clause has
|
||||||
// to keep its $1, $2, … in step with the order of the values — which SQLite's
|
// to keep its $1, $2, … in step with the order of the values. Handing out the placeholder and storing the value
|
||||||
// positional `?` did for free. Handing out the placeholder and storing the value
|
|
||||||
// in one call is what keeps them in step: a filter can be added, removed or
|
// in one call is what keeps them in step: a filter can be added, removed or
|
||||||
// reordered without renumbering anything by hand.
|
// reordered without renumbering anything by hand.
|
||||||
type sqlArgs struct{ vals []any }
|
type sqlArgs struct{ vals []any }
|
||||||
@@ -31,8 +31,7 @@ func (a *sqlArgs) add(v any) string {
|
|||||||
|
|
||||||
// addList stores every value and returns their placeholders as "$1, $2, …",
|
// addList stores every value and returns their placeholders as "$1, $2, …",
|
||||||
// ready to drop into an IN (…) clause. Returns an empty string for no values,
|
// ready to drop into an IN (…) clause. Returns an empty string for no values,
|
||||||
// which no caller should reach: `IN ()` is a syntax error in Postgres as it was
|
// which no caller should reach: `IN ()` is a syntax error in Postgres, so callers check for an empty set before building the query.
|
||||||
// in SQLite, so callers check for an empty set before building the query.
|
|
||||||
func (a *sqlArgs) addList(vs []any) string {
|
func (a *sqlArgs) addList(vs []any) string {
|
||||||
parts := make([]string, len(vs))
|
parts := make([]string, len(vs))
|
||||||
for i, v := range vs {
|
for i, v := range vs {
|
||||||
@@ -45,7 +44,7 @@ func (a *sqlArgs) addList(vs []any) string {
|
|||||||
func (a *sqlArgs) all() []any { return a.vals }
|
func (a *sqlArgs) all() []any { return a.vals }
|
||||||
|
|
||||||
// nowEpoch is the SQL expression for "now, as unix seconds", matching how every
|
// nowEpoch is the SQL expression for "now, as unix seconds", matching how every
|
||||||
// timestamp in this schema is stored. SQLite spelled it unixepoch().
|
// timestamp in this schema is stored.
|
||||||
//
|
//
|
||||||
// FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to
|
// FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to
|
||||||
// bigint rounds half up, so a row written at .6 of a second would claim a
|
// bigint rounds half up, so a row written at .6 of a second would claim a
|
||||||
@@ -56,11 +55,9 @@ const nowEpoch = "FLOOR(EXTRACT(EPOCH FROM now()))::bigint"
|
|||||||
// isUniqueViolation reports whether err is a broken unique constraint, which
|
// isUniqueViolation reports whether err is a broken unique constraint, which
|
||||||
// callers turn into 409 Conflict rather than 500.
|
// callers turn into 409 Conflict rather than 500.
|
||||||
//
|
//
|
||||||
// Postgres reports it as SQLSTATE 23505 on a typed error; the SQLite driver this
|
// Postgres reports it as SQLSTATE 23505 on a typed error. Matching the code
|
||||||
// replaced only put "UNIQUE constraint failed" in the message, which is why the
|
// means a renamed constraint or a translated message cannot quietly turn a
|
||||||
// check used to be a substring match. Matching the code means a renamed
|
// conflict back into a 500.
|
||||||
// constraint or a translated message cannot quietly turn a conflict back into a
|
|
||||||
// 500.
|
|
||||||
func isUniqueViolation(err error) bool {
|
func isUniqueViolation(err error) bool {
|
||||||
var pgErr *pgconn.PgError
|
var pgErr *pgconn.PgError
|
||||||
return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation
|
return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation
|
||||||
@@ -72,6 +69,18 @@ func respond(w http.ResponseWriter, status int, v any) {
|
|||||||
json.NewEncoder(w).Encode(v)
|
json.NewEncoder(w).Encode(v)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// serverError answers 500 and logs why. The response stays opaque, so the log
|
||||||
|
// line is the only record of what failed.
|
||||||
|
func serverError(w http.ResponseWriter, r *http.Request, err error) {
|
||||||
|
// The route pattern, not the path: two routes carry a credential in it.
|
||||||
|
route := r.URL.Path
|
||||||
|
if rc := chi.RouteContext(r.Context()); rc != nil && rc.RoutePattern() != "" {
|
||||||
|
route = rc.RoutePattern()
|
||||||
|
}
|
||||||
|
log.Printf("%s %s: %v", strconv.Quote(r.Method), strconv.Quote(route), err)
|
||||||
|
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||||
|
}
|
||||||
|
|
||||||
// maxBodyBytes caps an ordinary JSON request body. 1 MiB is far more than any
|
// maxBodyBytes caps an ordinary JSON request body. 1 MiB is far more than any
|
||||||
// endpoint below needs — it exists so an unauthenticated caller (signup,
|
// endpoint below needs — it exists so an unauthenticated caller (signup,
|
||||||
// login, bootstrap) can't make the server buffer an arbitrarily large body
|
// login, bootstrap) can't make the server buffer an arbitrarily large body
|
||||||
|
|||||||
@@ -13,6 +13,52 @@ import (
|
|||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// handleListClusters answers GET /api/incidents/clusters: the distinct values of
|
||||||
|
// the origin label across the caller's incidents from the last 90 days, sorted,
|
||||||
|
// so the queue can offer them as a filter. Optional team_id narrows it to one
|
||||||
|
// team. Empty when nothing carries the label, which is how the UI knows to show
|
||||||
|
// no filter at all.
|
||||||
|
func handleListClusters(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
args := &sqlArgs{}
|
||||||
|
where := []string{
|
||||||
|
"team_id = ANY(" + args.add(callerTeamIDs(r.Context())) + ")",
|
||||||
|
"triggered_at >= " + args.add(time.Now().AddDate(0, 0, -90).Unix()),
|
||||||
|
}
|
||||||
|
if team := r.URL.Query().Get("team_id"); team != "" {
|
||||||
|
if n, err := strconv.ParseInt(team, 10, 64); err == nil {
|
||||||
|
where = append(where, "team_id = "+args.add(n))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
label := args.add(originLabel)
|
||||||
|
|
||||||
|
rows, err := db.QueryContext(r.Context(),
|
||||||
|
fmt.Sprintf("SELECT DISTINCT group_labels->>%[1]s AS v FROM incidents WHERE %[2]s AND group_labels->>%[1]s <> '' ORDER BY v LIMIT 200",
|
||||||
|
label, strings.Join(where, " AND ")),
|
||||||
|
args.all()...)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
clusters := []string{}
|
||||||
|
for rows.Next() {
|
||||||
|
var v string
|
||||||
|
if err := rows.Scan(&v); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
clusters = append(clusters, v)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, clusters)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
q := r.URL.Query()
|
q := r.URL.Query()
|
||||||
@@ -53,6 +99,12 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
if severity := q.Get("severity"); severity != "" {
|
if severity := q.Get("severity"); severity != "" {
|
||||||
where = append(where, "i.severity = "+args.add(severity))
|
where = append(where, "i.severity = "+args.add(severity))
|
||||||
}
|
}
|
||||||
|
// Where it came from: the value of the origin label (originLabel, by
|
||||||
|
// convention "cluster") among the incident's group labels. An incident has
|
||||||
|
// it only when the label is in Alertmanager's group_by.
|
||||||
|
if cluster := q.Get("cluster"); cluster != "" {
|
||||||
|
where = append(where, "i.group_labels->>"+args.add(originLabel)+" = "+args.add(cluster))
|
||||||
|
}
|
||||||
if assignee := q.Get("assigned_to"); assignee != "" {
|
if assignee := q.Get("assigned_to"); assignee != "" {
|
||||||
if n, err := strconv.ParseInt(assignee, 10, 64); err == nil {
|
if n, err := strconv.ParseInt(assignee, 10, 64); err == nil {
|
||||||
where = append(where, "i.assigned_to = "+args.add(n))
|
where = append(where, "i.assigned_to = "+args.add(n))
|
||||||
@@ -86,7 +138,7 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
incidentSelectFrom, strings.Join(where, " AND "), order, args.add(limit)),
|
incidentSelectFrom, strings.Join(where, " AND "), order, args.add(limit)),
|
||||||
args.all()...)
|
args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -95,13 +147,13 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
i, err := scanIncident(rows)
|
i, err := scanIncident(rows)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
incidents = append(incidents, i)
|
incidents = append(incidents, i)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, incidents)
|
respond(w, http.StatusOK, incidents)
|
||||||
@@ -120,35 +172,17 @@ func handleGetIncident(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if inc.Alerts, err = incidentAlerts(r, db, id); err != nil {
|
if inc.Alerts, err = incidentAlerts(r, db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, inc)
|
respond(w, http.StatusOK, inc)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func handleIncidentAlerts(db *sql.DB) http.HandlerFunc {
|
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
|
||||||
id, ok := incidentIDParam(w, r, db)
|
|
||||||
if !ok {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if !incidentExists(w, r, db, id) {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
alerts, err := incidentAlerts(r, db, id)
|
|
||||||
if err != nil {
|
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
|
||||||
return
|
|
||||||
}
|
|
||||||
respond(w, http.StatusOK, alerts)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
id, ok := incidentIDParam(w, r, db)
|
id, ok := incidentIDParam(w, r, db)
|
||||||
@@ -173,7 +207,7 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE e.incident_id = $1
|
WHERE e.incident_id = $1
|
||||||
ORDER BY e.created_at ASC, e.id ASC`, id)
|
ORDER BY e.created_at ASC, e.id ASC`, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -187,14 +221,14 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
|||||||
&e.ActorUserID, &e.ActorUsername,
|
&e.ActorUserID, &e.ActorUsername,
|
||||||
&e.ActorServiceAccountID, &e.ActorServiceAccountName,
|
&e.ActorServiceAccountID, &e.ActorServiceAccountName,
|
||||||
&e.AlertID, &e.Detail, &ts); err != nil {
|
&e.AlertID, &e.Detail, &ts); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||||
events = append(events, e)
|
events = append(events, e)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, events)
|
respond(w, http.StatusOK, events)
|
||||||
@@ -210,7 +244,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
|
|||||||
userID, saID := callerActorIDs(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
acked, err := acknowledgeIncidentAs(r.Context(), db, id, userID, saID)
|
acked, err := acknowledgeIncidentAs(r.Context(), db, id, userID, saID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !acked {
|
if !acked {
|
||||||
@@ -220,7 +254,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
|
|||||||
// (acknowledged) already holds.
|
// (acknowledged) already holds.
|
||||||
inc, err := fetchIncident(r.Context(), db, id)
|
inc, err := fetchIncident(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if inc.Status == "resolved" {
|
if inc.Status == "resolved" {
|
||||||
@@ -248,7 +282,7 @@ func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := logEvent(r.Context(), db, id, evUnacknowledged, userID, saID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evUnacknowledged, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
@@ -283,16 +317,16 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
// A person closing an incident is the clearest possible "I have this".
|
// A person closing an incident is the clearest possible "I have this".
|
||||||
if err := stopEscalation(r.Context(), db, id); err != nil {
|
if err := stopEscalation(r.Context(), db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := logEvent(r.Context(), db, id, evResolved, userID, saID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evResolved, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if req.Resolution != "" {
|
if req.Resolution != "" {
|
||||||
if err := logEvent(r.Context(), db, id, evResolutionNote, userID, saID, nil, &req.Resolution); err != nil {
|
if err := logEvent(r.Context(), db, id, evResolutionNote, userID, saID, nil, &req.Resolution); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -333,7 +367,7 @@ func handleIncidentAssign(db *sql.DB) http.HandlerFunc {
|
|||||||
// the actor_* columns.
|
// the actor_* columns.
|
||||||
actorUserID, actorSAID := callerActorIDs(r.Context())
|
actorUserID, actorSAID := callerActorIDs(r.Context())
|
||||||
if err := logAssignedEvent(r.Context(), db, id, req.UserID, actorUserID, actorSAID); err != nil {
|
if err := logAssignedEvent(r.Context(), db, id, req.UserID, actorUserID, actorSAID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
@@ -391,7 +425,7 @@ func handleIncidentSnooze(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
detail := until.UTC().Format(time.RFC3339)
|
detail := until.UTC().Format(time.RFC3339)
|
||||||
if err := logEvent(r.Context(), db, id, evSnoozed, userID, saID, nil, &detail); err != nil {
|
if err := logEvent(r.Context(), db, id, evSnoozed, userID, saID, nil, &detail); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
@@ -410,7 +444,7 @@ func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := logEvent(r.Context(), db, id, evUnsnoozed, userID, saID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evUnsnoozed, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
@@ -426,7 +460,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE incidents SET archived_at = "+nowEpoch+" WHERE id = $1", id)
|
"UPDATE incidents SET archived_at = "+nowEpoch+" WHERE id = $1", id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -435,7 +469,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
userID, saID := callerActorIDs(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
if err := logEvent(r.Context(), db, id, evArchived, userID, saID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evArchived, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
@@ -451,7 +485,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE incidents SET archived_at = NULL WHERE id = $1", id)
|
"UPDATE incidents SET archived_at = NULL WHERE id = $1", id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -460,7 +494,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
userID, saID := callerActorIDs(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
if err := logEvent(r.Context(), db, id, evUnarchived, userID, saID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evUnarchived, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
@@ -505,7 +539,7 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
|||||||
VALUES ($1, $2, $3, $4, $5, $6)
|
VALUES ($1, $2, $3, $4, $5, $6)
|
||||||
RETURNING id`, id, noteType, userID, saID, req.Content, now.Unix()).Scan(&eventID)
|
RETURNING id`, id, noteType, userID, saID, req.Content, now.Unix()).Scan(&eventID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -546,7 +580,7 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc {
|
|||||||
AND (user_id = $5 OR service_account_id = $6)`,
|
AND (user_id = $5 OR service_account_id = $6)`,
|
||||||
eventID, id, evNote, evResolutionNote, userID, saID)
|
eventID, id, evNote, evResolutionNote, userID, saID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -600,7 +634,7 @@ func incidentExists(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64
|
|||||||
func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64, query string, args ...any) bool {
|
func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64, query string, args ...any) bool {
|
||||||
res, err := db.ExecContext(r.Context(), query, args...)
|
res, err := db.ExecContext(r.Context(), query, args...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n > 0 {
|
if n, _ := res.RowsAffected(); n > 0 {
|
||||||
@@ -616,7 +650,7 @@ func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id i
|
|||||||
func respondIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) {
|
func respondIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) {
|
||||||
inc, err := fetchIncident(r.Context(), db, id)
|
inc, err := fetchIncident(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, inc)
|
respond(w, http.StatusOK, inc)
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
@@ -793,8 +794,7 @@ func TestStats_Incidents(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// An empty window is a report of zero, not a failure. SUM over no rows is NULL
|
// An empty window is a report of zero, not a failure. SUM over no rows is NULL
|
||||||
// in Postgres as it was in SQLite, and that used to come back as a 500 the
|
// in Postgres, and that used to come back as a 500 the moment every incident was archived — the state a quiet installation settles
|
||||||
// moment every incident was archived — the state a quiet installation settles
|
|
||||||
// into.
|
// into.
|
||||||
func TestStats_IncidentsEmptyWindowIsZeroNotAnError(t *testing.T) {
|
func TestStats_IncidentsEmptyWindowIsZeroNotAnError(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
@@ -972,3 +972,37 @@ func TestServiceAccount_AssignArchiveUnarchiveRecordActor(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The queue can be narrowed to one cluster, and the distinct clusters are
|
||||||
|
// offered so the filter has something to list.
|
||||||
|
func TestIncidents_FilterByCluster(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
for _, c := range []string{"prod-eu", "prod-us"} {
|
||||||
|
fireGroupedAs(t, s, "fp-"+c,
|
||||||
|
map[string]string{"alertname": "PodRestarting", "cluster": c, "namespace": "n"},
|
||||||
|
`{}:{alertname="PodRestarting",cluster="`+c+`",namespace="n"}`)
|
||||||
|
}
|
||||||
|
// A cluster-less incident exists too, and must never match a cluster filter.
|
||||||
|
fireGroupedAs(t, s, "fp-none", map[string]string{"alertname": "DiskFull"}, `{}:{alertname="DiskFull"}`)
|
||||||
|
|
||||||
|
if got := len(listIncidents(t, s, "")); got != 3 {
|
||||||
|
t.Fatalf("expected 3 open incidents, got %d", got)
|
||||||
|
}
|
||||||
|
eu := listIncidents(t, s, "?cluster=prod-eu")
|
||||||
|
if len(eu) != 1 {
|
||||||
|
t.Fatalf("expected 1 incident for prod-eu, got %d", len(eu))
|
||||||
|
}
|
||||||
|
if labels, _ := eu[0]["group_labels"].(map[string]any); labels["cluster"] != "prod-eu" {
|
||||||
|
t.Errorf("filtered to the wrong cluster: %v", eu[0]["group_labels"])
|
||||||
|
}
|
||||||
|
if got := len(listIncidents(t, s, "?cluster=nowhere")); got != 0 {
|
||||||
|
t.Errorf("an unknown cluster should match nothing, got %d", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
var clusters []string
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/incidents/clusters", nil), &clusters)
|
||||||
|
if want := "prod-eu,prod-us"; strings.Join(clusters, ",") != want {
|
||||||
|
t.Errorf("clusters = %v, want %s", clusters, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -5,10 +5,14 @@ import (
|
|||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"github.com/go-chi/chi/v5"
|
||||||
|
"github.com/go-chi/chi/v5/middleware"
|
||||||
|
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
)
|
)
|
||||||
@@ -106,7 +110,7 @@ func securityHeaders(publicURL string) func(http.Handler) http.Handler {
|
|||||||
//
|
//
|
||||||
// 403 and not 404: the route exists and the caller is authenticated, they are
|
// 403 and not 404: the route exists and the caller is authenticated, they are
|
||||||
// simply not allowed. Hiding the endpoint would buy nothing — every one of them
|
// simply not allowed. Hiding the endpoint would buy nothing — every one of them
|
||||||
// is in the README.
|
// is in docs/api.md.
|
||||||
func AdminOnly(next http.Handler) http.Handler {
|
func AdminOnly(next http.Handler) http.Handler {
|
||||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
caller, ok := userFromContext(r.Context())
|
caller, ok := userFromContext(r.Context())
|
||||||
@@ -141,19 +145,23 @@ func requireSelfOrAdmin(w http.ResponseWriter, r *http.Request, targetID int64)
|
|||||||
// resolve and has to be caught afterwards.
|
// resolve and has to be caught afterwards.
|
||||||
func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) {
|
func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) {
|
||||||
var keyID, userID int64
|
var keyID, userID int64
|
||||||
|
var lastUsed sql.NullInt64
|
||||||
err := db.QueryRowContext(ctx,
|
err := db.QueryRowContext(ctx,
|
||||||
`SELECT id, user_id FROM api_keys
|
`SELECT id, user_id, last_used_at FROM api_keys
|
||||||
WHERE key_hash = $1 AND (expires_at IS NULL OR expires_at > $2)`,
|
WHERE key_hash = $1 AND (expires_at IS NULL OR expires_at > $2)`,
|
||||||
hashToken(token), time.Now().Unix(),
|
hashToken(token), time.Now().Unix(),
|
||||||
).Scan(&keyID, &userID)
|
).Scan(&keyID, &userID, &lastUsed)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, false
|
return 0, false
|
||||||
}
|
}
|
||||||
|
|
||||||
// best-effort; don't fail the request if this update fails
|
// best-effort; don't fail the request if this update fails. Throttled like
|
||||||
db.ExecContext(ctx,
|
// the session expiry, so a polling client does not write a row per request.
|
||||||
"UPDATE api_keys SET last_used_at = $1 WHERE id = $2",
|
if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery {
|
||||||
time.Now().Unix(), keyID)
|
db.ExecContext(ctx,
|
||||||
|
"UPDATE api_keys SET last_used_at = $1 WHERE id = $2",
|
||||||
|
now.Unix(), keyID)
|
||||||
|
}
|
||||||
return userID, true
|
return userID, true
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -206,7 +214,7 @@ func serveAs(w http.ResponseWriter, r *http.Request, next http.Handler, db *sql.
|
|||||||
// table with one row per membership.
|
// table with one row per membership.
|
||||||
teams, err := callerMemberships(r.Context(), db, userID)
|
teams, err := callerMemberships(r.Context(), db, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -222,10 +230,8 @@ func hashToken(token string) string {
|
|||||||
return hex.EncodeToString(h[:])
|
return hex.EncodeToString(h[:])
|
||||||
}
|
}
|
||||||
|
|
||||||
// userFromContext is a thin compatibility wrapper over Caller.AsHuman(), so
|
// userFromContext returns the human behind the request, or false for a service
|
||||||
// every call site written before the Caller abstraction (alerts.go,
|
// account: Caller.AsHuman() on the request's Caller.
|
||||||
// incidents.go, schedule.go, stats.go, and more) needs no change and keeps
|
|
||||||
// its exact existing behavior.
|
|
||||||
func userFromContext(ctx context.Context) (models.User, bool) {
|
func userFromContext(ctx context.Context) (models.User, bool) {
|
||||||
c, _ := callerFromContext(ctx)
|
c, _ := callerFromContext(ctx)
|
||||||
return c.AsHuman()
|
return c.AsHuman()
|
||||||
@@ -245,13 +251,13 @@ type serviceAccountPrincipal struct {
|
|||||||
func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAccountPrincipal, bool) {
|
func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAccountPrincipal, bool) {
|
||||||
var sa serviceAccountPrincipal
|
var sa serviceAccountPrincipal
|
||||||
var keyID int64
|
var keyID int64
|
||||||
var teamID sql.NullInt64
|
var teamID, lastUsed sql.NullInt64
|
||||||
err := db.QueryRowContext(ctx, `
|
err := db.QueryRowContext(ctx, `
|
||||||
SELECT k.id, a.id, a.name, a.scope, a.team_id
|
SELECT k.id, a.id, a.name, a.scope, a.team_id, k.last_used_at
|
||||||
FROM service_account_keys k
|
FROM service_account_keys k
|
||||||
JOIN service_accounts a ON a.id = k.service_account_id
|
JOIN service_accounts a ON a.id = k.service_account_id
|
||||||
WHERE k.key_hash = $1`, hashToken(token),
|
WHERE k.key_hash = $1`, hashToken(token),
|
||||||
).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID)
|
).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID, &lastUsed)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return serviceAccountPrincipal{}, false
|
return serviceAccountPrincipal{}, false
|
||||||
}
|
}
|
||||||
@@ -260,9 +266,11 @@ func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAc
|
|||||||
}
|
}
|
||||||
|
|
||||||
// best-effort; don't fail the request if this update fails
|
// best-effort; don't fail the request if this update fails
|
||||||
db.ExecContext(ctx,
|
if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery {
|
||||||
"UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2",
|
db.ExecContext(ctx,
|
||||||
time.Now().Unix(), keyID)
|
"UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2",
|
||||||
|
now.Unix(), keyID)
|
||||||
|
}
|
||||||
return sa, true
|
return sa, true
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -286,10 +294,8 @@ func serveAsServiceAccount(w http.ResponseWriter, r *http.Request, next http.Han
|
|||||||
next.ServeHTTP(w, r.WithContext(ctx))
|
next.ServeHTTP(w, r.WithContext(ctx))
|
||||||
}
|
}
|
||||||
|
|
||||||
// isInstanceServiceAccount is a thin compatibility wrapper over
|
// isInstanceServiceAccount is Caller.IsInstanceServiceAccount() on the
|
||||||
// Caller.IsInstanceServiceAccount(), for call sites outside this package's
|
// request's Caller.
|
||||||
// core predicates (handleCreateTeam, handleCreateServiceAccount) that
|
|
||||||
// needed this exact, narrow check before the Caller abstraction existed.
|
|
||||||
func isInstanceServiceAccount(ctx context.Context) bool {
|
func isInstanceServiceAccount(ctx context.Context) bool {
|
||||||
c, _ := callerFromContext(ctx)
|
c, _ := callerFromContext(ctx)
|
||||||
return c.IsInstanceServiceAccount()
|
return c.IsInstanceServiceAccount()
|
||||||
@@ -397,6 +403,13 @@ func requireTeamOwner(w http.ResponseWriter, r *http.Request, teamID int64) bool
|
|||||||
if ok && role == models.RoleOwner {
|
if ok && role == models.RoleOwner {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
// The operator's instance-scoped account manages every team's
|
||||||
|
// configuration, which is what lets it use one credential instead of
|
||||||
|
// minting one per team. This is owner reach only: it does not make the
|
||||||
|
// account a member, so it still reads no team's incidents.
|
||||||
|
if c, _ := callerFromContext(r.Context()); c.IsInstanceServiceAccount() {
|
||||||
|
return true
|
||||||
|
}
|
||||||
if caller, _ := userFromContext(r.Context()); caller.IsAdmin {
|
if caller, _ := userFromContext(r.Context()); caller.IsAdmin {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
@@ -414,3 +427,26 @@ func sessionFromContext(ctx context.Context) (int64, bool) {
|
|||||||
id, ok := ctx.Value(ctxSession).(int64)
|
id, ok := ctx.Value(ctxSession).(int64)
|
||||||
return id, ok
|
return id, ok
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// requestLogger logs one line per request with the matched route pattern in
|
||||||
|
// place of the URL path. Two routes carry a credential in the path (the
|
||||||
|
// integration key and the ack token), and chi's stock logger would write it to
|
||||||
|
// the log verbatim.
|
||||||
|
func requestLogger(next http.Handler) http.Handler {
|
||||||
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
start := time.Now()
|
||||||
|
ww := middleware.NewWrapResponseWriter(w, r.ProtoMajor)
|
||||||
|
next.ServeHTTP(ww, r)
|
||||||
|
route := "unmatched"
|
||||||
|
if rc := chi.RouteContext(r.Context()); rc != nil {
|
||||||
|
if p := rc.RoutePattern(); p != "" {
|
||||||
|
route = p
|
||||||
|
}
|
||||||
|
}
|
||||||
|
status := ww.Status()
|
||||||
|
if status == 0 {
|
||||||
|
status = http.StatusOK
|
||||||
|
}
|
||||||
|
log.Printf("%q %q %d %dB %s", r.Method, route, status, ww.BytesWritten(), time.Since(start).Round(time.Millisecond)) // #nosec G706 -- method and route are %q-quoted, the route is a registered pattern, the rest are numbers
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"regexp"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
@@ -409,9 +410,11 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify
|
|||||||
strings.TrimSuffix(cfg.PublicURL, "/"), inc.ID)
|
strings.TrimSuffix(cfg.PublicURL, "/"), inc.ID)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
title := pageTitle(inc)
|
||||||
|
|
||||||
switch n.kind {
|
switch n.kind {
|
||||||
case notifyResolved:
|
case notifyResolved:
|
||||||
msg.Title = "Resolved: " + inc.Title
|
msg.Title = "Resolved: " + title
|
||||||
msg.Message = "All alerts stopped firing after " +
|
msg.Message = "All alerts stopped firing after " +
|
||||||
humanDuration(time.Since(inc.TriggeredAt))
|
humanDuration(time.Since(inc.TriggeredAt))
|
||||||
msg.Priority = ntfyPriorityLow
|
msg.Priority = ntfyPriorityLow
|
||||||
@@ -419,9 +422,9 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify
|
|||||||
return msg
|
return msg
|
||||||
|
|
||||||
case notifyReminder:
|
case notifyReminder:
|
||||||
msg.Title = "Still unacknowledged: " + inc.Title
|
msg.Title = "Still unacknowledged: " + title
|
||||||
default:
|
default:
|
||||||
msg.Title = inc.Title
|
msg.Title = title
|
||||||
}
|
}
|
||||||
|
|
||||||
severity := derefString(inc.Severity)
|
severity := derefString(inc.Severity)
|
||||||
@@ -443,6 +446,48 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify
|
|||||||
return msg
|
return msg
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// originLabel is the label that says where an alert came from, for a team with
|
||||||
|
// several Kubernetes clusters behind it. It comes from Prometheus's
|
||||||
|
// externalLabels and reaches an incident through Alertmanager's group_by; the
|
||||||
|
// web UI reads the same label, and docs/incidents.md ("Several clusters, one team")
|
||||||
|
// explains how to set it up.
|
||||||
|
const originLabel = "cluster"
|
||||||
|
|
||||||
|
// pageTitle is the incident's title for a notification. A phone's lock screen
|
||||||
|
// cuts a long title off at the end, and the incident title puts the grouping
|
||||||
|
// labels there, so the cluster would be the first thing lost. When the incident
|
||||||
|
// has an origin it leads instead, "[prod-eu] PodRestarting (namespace=foo)", and
|
||||||
|
// is dropped from the parenthesis so it is not said twice. A title that is not
|
||||||
|
// in incidentTitle's "name (k=v, k=v)" shape keeps its text and gains the prefix.
|
||||||
|
func pageTitle(inc models.Incident) string {
|
||||||
|
origin := inc.GroupLabels[originLabel]
|
||||||
|
if origin == "" {
|
||||||
|
return inc.Title
|
||||||
|
}
|
||||||
|
return "[" + origin + "] " + titleWithoutLabel(inc.Title, originLabel, origin)
|
||||||
|
}
|
||||||
|
|
||||||
|
var titleShape = regexp.MustCompile(`(?s)^(.*?) \((.*)\)$`)
|
||||||
|
|
||||||
|
// titleWithoutLabel removes "key=value" from the parenthesised tail of a title
|
||||||
|
// built by incidentTitle, and the parentheses with it if nothing else is left.
|
||||||
|
func titleWithoutLabel(title, key, value string) string {
|
||||||
|
m := titleShape.FindStringSubmatch(title)
|
||||||
|
if m == nil {
|
||||||
|
return title
|
||||||
|
}
|
||||||
|
var rest []string
|
||||||
|
for _, part := range strings.Split(m[2], ", ") {
|
||||||
|
if part != key+"="+value {
|
||||||
|
rest = append(rest, part)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(rest) == 0 {
|
||||||
|
return m[1]
|
||||||
|
}
|
||||||
|
return m[1] + " (" + strings.Join(rest, ", ") + ")"
|
||||||
|
}
|
||||||
|
|
||||||
// ntfy's priority scale. Max is the one that overrides the phone's quiet
|
// ntfy's priority scale. Max is the one that overrides the phone's quiet
|
||||||
// settings, which is the whole point of paging on critical.
|
// settings, which is the whole point of paging on critical.
|
||||||
const (
|
const (
|
||||||
|
|||||||
@@ -62,7 +62,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID)
|
acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !acked {
|
if !acked {
|
||||||
@@ -73,7 +73,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc {
|
|||||||
// ntfy show a success toast rather than a failure.
|
// ntfy show a success toast rather than a failure.
|
||||||
inc, err := fetchIncident(r.Context(), db, incidentID)
|
inc, err := fetchIncident(r.Context(), db, incidentID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, map[string]any{
|
respond(w, http.StatusOK, map[string]any{
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
package api_test
|
package api_test
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"fmt"
|
"fmt"
|
||||||
@@ -164,10 +165,91 @@ func fireCritical(t *testing.T, s *ts) {
|
|||||||
}, "{}:{alertname=\"DiskFull\"}")
|
}, "{}:{alertname=\"DiskFull\"}")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// fireGrouped posts one critical alert whose Alertmanager group carries the
|
||||||
|
// given labels, the way group_by puts them on the webhook.
|
||||||
|
func fireGrouped(t *testing.T, s *ts, groupLabels map[string]string, groupKey string) {
|
||||||
|
t.Helper()
|
||||||
|
fireGroupedAs(t, s, "fp-grouped", groupLabels, groupKey)
|
||||||
|
}
|
||||||
|
|
||||||
|
// fireGroupedAs is fireGrouped with its own alert fingerprint, for a test that
|
||||||
|
// needs several alerts open at once.
|
||||||
|
func fireGroupedAs(t *testing.T, s *ts, fingerprint string, groupLabels map[string]string, groupKey string) {
|
||||||
|
t.Helper()
|
||||||
|
payload := map[string]any{
|
||||||
|
"version": "4", "status": "firing", "groupKey": groupKey, "groupLabels": groupLabels,
|
||||||
|
"alerts": []map[string]any{amAlert(fingerprint, "PodRestarting", "firing",
|
||||||
|
"2026-05-20T10:00:00Z", zeroTime, map[string]string{"severity": "critical"})},
|
||||||
|
}
|
||||||
|
data, _ := json.Marshal(payload)
|
||||||
|
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||||
|
"application/json", bytes.NewReader(data))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("post webhook: %v", err)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("webhook returned %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Delivery
|
// Delivery
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// A phone cuts a long title off at the end, and the incident title keeps the
|
||||||
|
// grouping labels there, so the cluster leads the page instead.
|
||||||
|
func TestNotify_ClusterLeadsTheTitle(t *testing.T) {
|
||||||
|
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||||
|
|
||||||
|
fireGrouped(t, s, map[string]string{
|
||||||
|
"alertname": "PodRestarting", "cluster": "prod-eu", "namespace": "shop",
|
||||||
|
}, `{}:{alertname="PodRestarting",cluster="prod-eu",namespace="shop"}`)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
msgs := f.messages()
|
||||||
|
if len(msgs) != 1 {
|
||||||
|
t.Fatalf("expected 1 push, got %d", len(msgs))
|
||||||
|
}
|
||||||
|
if want := "[prod-eu] PodRestarting (namespace=shop)"; msgs[0].Title != want {
|
||||||
|
t.Errorf("title = %q, want %q", msgs[0].Title, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Only the cluster is the whole grouping: no parenthesis is left behind.
|
||||||
|
func TestNotify_ClusterAloneLeavesNoParenthesis(t *testing.T) {
|
||||||
|
s, f := notifyTS(t, api.NotifyConfig{})
|
||||||
|
|
||||||
|
fireGrouped(t, s, map[string]string{"alertname": "PodRestarting", "cluster": "prod-eu"},
|
||||||
|
`{}:{alertname="PodRestarting",cluster="prod-eu"}`)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
msgs := f.messages()
|
||||||
|
if len(msgs) != 1 {
|
||||||
|
t.Fatalf("expected 1 push, got %d", len(msgs))
|
||||||
|
}
|
||||||
|
if want := "[prod-eu] PodRestarting"; msgs[0].Title != want {
|
||||||
|
t.Errorf("title = %q, want %q", msgs[0].Title, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nothing changes for a team whose alerts have no cluster label.
|
||||||
|
func TestNotify_NoClusterKeepsTheTitle(t *testing.T) {
|
||||||
|
s, f := notifyTS(t, api.NotifyConfig{})
|
||||||
|
|
||||||
|
fireGrouped(t, s, map[string]string{"alertname": "PodRestarting", "namespace": "shop"},
|
||||||
|
`{}:{alertname="PodRestarting",namespace="shop"}`)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
msgs := f.messages()
|
||||||
|
if len(msgs) != 1 {
|
||||||
|
t.Fatalf("expected 1 push, got %d", len(msgs))
|
||||||
|
}
|
||||||
|
if want := "PodRestarting (namespace=shop)"; msgs[0].Title != want {
|
||||||
|
t.Errorf("title = %q, want %q", msgs[0].Title, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestNotify_TriggeredIncidentPagesOnCall(t *testing.T) {
|
func TestNotify_TriggeredIncidentPagesOnCall(t *testing.T) {
|
||||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||||
|
|
||||||
|
|||||||
@@ -130,12 +130,12 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub
|
|||||||
|
|
||||||
state, stateHash, err := randomToken()
|
state, stateHash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
nonce, _, err := randomToken()
|
nonce, _, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
verifier := oidc.NewVerifier()
|
verifier := oidc.NewVerifier()
|
||||||
@@ -149,7 +149,7 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub
|
|||||||
INSERT INTO oidc_logins (state_hash, nonce, pkce_verifier, next, expires_at)
|
INSERT INTO oidc_logins (state_hash, nonce, pkce_verifier, next, expires_at)
|
||||||
VALUES ($1, $2, $3, $4, $5)`,
|
VALUES ($1, $2, $3, $4, $5)`,
|
||||||
stateHash, nonce, verifier, next, now.Add(oidcLoginTTL).Unix()); err != nil {
|
stateHash, nonce, verifier, next, now.Add(oidcLoginTTL).Unix()); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ func handleGetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
|
|||||||
"SELECT COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams WHERE id = $1",
|
"SELECT COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams WHERE id = $1",
|
||||||
teamID).Scan(&g.MemberGroup, &g.OwnerGroup)
|
teamID).Scan(&g.MemberGroup, &g.OwnerGroup)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, g)
|
respond(w, http.StatusOK, g)
|
||||||
@@ -66,7 +66,7 @@ func handleSetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
|
|||||||
oidc_owner_group = NULLIF($2, '')
|
oidc_owner_group = NULLIF($2, '')
|
||||||
WHERE id = $3`,
|
WHERE id = $3`,
|
||||||
req.MemberGroup, req.OwnerGroup, teamID); err != nil {
|
req.MemberGroup, req.OwnerGroup, teamID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
|||||||
@@ -0,0 +1,72 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
// operatorAccountName is the instance-scoped service account the operator
|
||||||
|
// key belongs to.
|
||||||
|
operatorAccountName = "terdut-operator"
|
||||||
|
|
||||||
|
// operatorKeyName names the one key SeedOperatorKey manages on it, so a
|
||||||
|
// rotation replaces that key and leaves any others alone.
|
||||||
|
operatorKeyName = "seed"
|
||||||
|
)
|
||||||
|
|
||||||
|
// SeedOperatorKey makes key the operator account's credential: it creates the
|
||||||
|
// instance-scoped service account if needed and replaces its "seed" key with
|
||||||
|
// this one. Idempotent, so every replica can run it at every start, and a
|
||||||
|
// rotated key simply wins on the next restart.
|
||||||
|
//
|
||||||
|
// The key is hashed like any other, so only the caller that generated it ever
|
||||||
|
// holds the raw value. An empty key does nothing.
|
||||||
|
func SeedOperatorKey(ctx context.Context, db *sql.DB, key string) error {
|
||||||
|
if key == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
tx, err := db.BeginTx(ctx, nil)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: %w", err)
|
||||||
|
}
|
||||||
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
|
// Serialise replicas starting together; transaction-scoped, so it needs no
|
||||||
|
// explicit release.
|
||||||
|
if _, err := tx.ExecContext(ctx, "SELECT pg_advisory_xact_lock($1)", operatorKeyLockKey); err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: lock: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var accountID int64
|
||||||
|
err = tx.QueryRowContext(ctx,
|
||||||
|
"SELECT id FROM service_accounts WHERE name = $1 AND scope = $2",
|
||||||
|
operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID)
|
||||||
|
if err == sql.ErrNoRows {
|
||||||
|
err = tx.QueryRowContext(ctx,
|
||||||
|
"INSERT INTO service_accounts (name, scope) VALUES ($1, $2) RETURNING id",
|
||||||
|
operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: account: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := tx.ExecContext(ctx,
|
||||||
|
"DELETE FROM service_account_keys WHERE service_account_id = $1 AND name = $2",
|
||||||
|
accountID, operatorKeyName); err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: drop old key: %w", err)
|
||||||
|
}
|
||||||
|
if _, err := tx.ExecContext(ctx,
|
||||||
|
"INSERT INTO service_account_keys (service_account_id, key_hash, name) VALUES ($1, $2, $3)",
|
||||||
|
accountID, hashToken(key), operatorKeyName); err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: store key: %w", err)
|
||||||
|
}
|
||||||
|
return tx.Commit()
|
||||||
|
}
|
||||||
|
|
||||||
|
// operatorKeyLockKey is the transaction-scoped advisory lock SeedOperatorKey
|
||||||
|
// holds; distinct from the other lock keys in this package.
|
||||||
|
const operatorKeyLockKey int64 = 7265_0010
|
||||||
@@ -0,0 +1,134 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
operatorKeyOne = "tdsa_operator-key-number-one-0123456789"
|
||||||
|
operatorKeyTwo = "tdsa_operator-key-number-two-0123456789"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestSeedOperatorKey_AuthenticatesAsInstanceAccount(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", map[string]string{"name": "seeded"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("seeded key should create a team, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSeedOperatorKey_RotationReplacesAndIsIdempotent(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
for _, key := range []string{operatorKeyOne, operatorKeyOne, operatorKeyTwo} {
|
||||||
|
if err := api.SeedOperatorKey(ctx, s.db, key); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
old := s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/teams", nil)
|
||||||
|
old.Body.Close()
|
||||||
|
if old.StatusCode != http.StatusUnauthorized {
|
||||||
|
t.Errorf("rotated-out key should be refused, got %d", old.StatusCode)
|
||||||
|
}
|
||||||
|
cur := s.reqAs(t, operatorKeyTwo, http.MethodGet, "/api/teams", nil)
|
||||||
|
cur.Body.Close()
|
||||||
|
if cur.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("current key should work, got %d", cur.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
var accounts, keys int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM service_accounts WHERE name = 'terdut-operator'").Scan(&accounts)
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM service_account_keys").Scan(&keys)
|
||||||
|
if accounts != 1 || keys != 1 {
|
||||||
|
t.Errorf("expected one account and one key, got %d and %d", accounts, keys)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSeedOperatorKey_EmptyKeyDoesNothing(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, ""); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var n int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM service_accounts").Scan(&n)
|
||||||
|
if n != 0 {
|
||||||
|
t.Errorf("expected no service account, got %d", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The instance account configures a team it did not create, which is what lets
|
||||||
|
// the operator hold one credential instead of one per team, yet it is not a
|
||||||
|
// member and so reads none of the team's incidents.
|
||||||
|
func TestInstanceAccount_ActsAsOwnerOfAnyTeamButIsNoMember(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
other := newTeam(t, s, "other")
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
rename := s.reqAs(t, operatorKeyOne, http.MethodPut, "/api/teams/"+id64(other.id), map[string]string{"name": "renamed"})
|
||||||
|
rename.Body.Close()
|
||||||
|
if rename.StatusCode >= 300 {
|
||||||
|
t.Errorf("instance account should rename any team, got %d", rename.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Not a member: the team's queue is not visible to it.
|
||||||
|
var queue []map[string]any
|
||||||
|
decode(t, s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/incidents", nil), &queue)
|
||||||
|
if len(queue) != 0 {
|
||||||
|
t.Errorf("instance account should see no incidents, got %v", queue)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// external_id lets automation find its own team again after a crash, without
|
||||||
|
// trusting a display name.
|
||||||
|
func TestCreateTeam_ExternalIDIsIdempotentAndInstanceOnly(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
create := func(name string) (int, map[string]any) {
|
||||||
|
resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams",
|
||||||
|
map[string]string{"name": name, "external_id": "ns/platform"})
|
||||||
|
var out map[string]any
|
||||||
|
_ = json.NewDecoder(resp.Body).Decode(&out)
|
||||||
|
resp.Body.Close()
|
||||||
|
return resp.StatusCode, out
|
||||||
|
}
|
||||||
|
|
||||||
|
code, first := create("Platform")
|
||||||
|
if code != http.StatusCreated {
|
||||||
|
t.Fatalf("first create: %d", code)
|
||||||
|
}
|
||||||
|
// Same identity, even under a new display name: the same team comes back.
|
||||||
|
code, again := create("Platform renamed")
|
||||||
|
if code != http.StatusOK || again["id"] != first["id"] {
|
||||||
|
t.Errorf("repeat with the same external_id: want 200 and team %v, got %d %v", first["id"], code, again)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A different identity cannot take the name.
|
||||||
|
resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams",
|
||||||
|
map[string]string{"name": "Platform", "external_id": "other/platform"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("taken name under another external_id: want 409, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A person cannot set one.
|
||||||
|
resp = s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Mine", "external_id": "x/y"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("a user setting external_id: want 403, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -23,12 +23,13 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
|
|||||||
// One limiter each, both process-wide for the life of the router: login
|
// One limiter each, both process-wide for the life of the router: login
|
||||||
// counts failed passwords, sign-up counts account creation, and mixing the
|
// counts failed passwords, sign-up counts account creation, and mixing the
|
||||||
// two would let a burst of sign-ups lock somebody out of logging in.
|
// two would let a burst of sign-ups lock somebody out of logging in.
|
||||||
|
trustedProxies.Store(int64(cfg.TrustedProxies))
|
||||||
loginLimit := newLoginLimiter(db)
|
loginLimit := newLoginLimiter(db)
|
||||||
signupLimiter := newLoginLimiter(db)
|
signupLimiter := newLoginLimiter(db)
|
||||||
oidcLimit := newLoginLimiter(db)
|
oidcLimit := newLoginLimiter(db)
|
||||||
|
|
||||||
r := chi.NewRouter()
|
r := chi.NewRouter()
|
||||||
r.Use(middleware.Logger)
|
r.Use(requestLogger)
|
||||||
r.Use(middleware.Recoverer)
|
r.Use(middleware.Recoverer)
|
||||||
r.Use(securityHeaders(notify.PublicURL))
|
r.Use(securityHeaders(notify.PublicURL))
|
||||||
|
|
||||||
@@ -48,11 +49,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
|
|||||||
|
|
||||||
// Alert ingestion. The key in the path says both that the sender may post
|
// Alert ingestion. The key in the path says both that the sender may post
|
||||||
// and which team the alerts belong to, which is why it needs no session.
|
// and which team the alerts belong to, which is why it needs no session.
|
||||||
//
|
// This is the only way in.
|
||||||
// This is the only way in. The pre-teams /api/alertmanager/webhook, which
|
|
||||||
// took no credential at all, was removed in v0.13.0 once the cluster's
|
|
||||||
// Alertmanager had moved onto a key; a sender still posting there gets the
|
|
||||||
// JSON 404 every unknown /api path gets.
|
|
||||||
r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify))
|
r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify))
|
||||||
|
|
||||||
// Signing up. Both are unauthenticated by necessity: the caller has no
|
// Signing up. Both are unauthenticated by necessity: the caller has no
|
||||||
@@ -116,9 +113,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
|
|||||||
r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db))
|
r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db))
|
||||||
r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db))
|
r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db))
|
||||||
|
|
||||||
// Administration: who exists, and who is an administrator. Until #3
|
// Administration: who exists, and who is an administrator.
|
||||||
// these were open to any authenticated caller, which meant every user
|
|
||||||
// could delete every other one.
|
|
||||||
r.Group(func(r chi.Router) {
|
r.Group(func(r chi.Router) {
|
||||||
r.Use(AdminOnly)
|
r.Use(AdminOnly)
|
||||||
|
|
||||||
@@ -145,8 +140,8 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
|
|||||||
r.Get("/api/alerts/{id}", handleGetAlert(db))
|
r.Get("/api/alerts/{id}", handleGetAlert(db))
|
||||||
|
|
||||||
r.Get("/api/incidents", handleListIncidents(db))
|
r.Get("/api/incidents", handleListIncidents(db))
|
||||||
|
r.Get("/api/incidents/clusters", handleListClusters(db))
|
||||||
r.Get("/api/incidents/{id}", handleGetIncident(db))
|
r.Get("/api/incidents/{id}", handleGetIncident(db))
|
||||||
r.Get("/api/incidents/{id}/alerts", handleIncidentAlerts(db))
|
|
||||||
r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db))
|
r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db))
|
||||||
r.Get("/api/incidents/{id}/similar", handleIncidentSimilar(db))
|
r.Get("/api/incidents/{id}/similar", handleIncidentSimilar(db))
|
||||||
r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db))
|
r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db))
|
||||||
|
|||||||
@@ -69,7 +69,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
// be, so the delete and the insert share one transaction.
|
// be, so the delete and the insert share one transaction.
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback()
|
defer tx.Rollback()
|
||||||
@@ -79,7 +79,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
"DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
||||||
teamID, d); err != nil {
|
teamID, d); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -91,12 +91,12 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
errResp("date already assigned: "+d+" (pass replace to take it)"))
|
errResp("date already assigned: "+d+" (pass replace to take it)"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -107,7 +107,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1])
|
all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1])
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
created := []models.ScheduleEntry{}
|
created := []models.ScheduleEntry{}
|
||||||
@@ -147,7 +147,7 @@ func handleListSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
entries, err := scheduleRange(r.Context(), db, teamID, from, to)
|
entries, err := scheduleRange(r.Context(), db, teamID, from, to)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, entries)
|
respond(w, http.StatusOK, entries)
|
||||||
@@ -171,7 +171,7 @@ func handleDeleteSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID)
|
"DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -197,7 +197,7 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE s.date = $1 AND s.team_id = ANY($2)
|
WHERE s.date = $1 AND s.team_id = ANY($2)
|
||||||
ORDER BY t.name`, today, callerTeamIDs(r.Context()))
|
ORDER BY t.name`, today, callerTeamIDs(r.Context()))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -207,14 +207,14 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
var e models.ScheduleEntry
|
var e models.ScheduleEntry
|
||||||
var ts int64
|
var ts int64
|
||||||
if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
|
if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||||
entries = append(entries, e)
|
entries = append(entries, e)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, entries)
|
respond(w, http.StatusOK, entries)
|
||||||
|
|||||||
@@ -50,6 +50,9 @@ func callerIsAdmin(ctx context.Context) bool {
|
|||||||
// writing a response: callers here need to combine it with other ways of
|
// writing a response: callers here need to combine it with other ways of
|
||||||
// being allowed, not stop at the first no.
|
// being allowed, not stop at the first no.
|
||||||
func callerOwnsTeam(ctx context.Context, teamID int64) bool {
|
func callerOwnsTeam(ctx context.Context, teamID int64) bool {
|
||||||
|
if c, _ := callerFromContext(ctx); c.IsInstanceServiceAccount() {
|
||||||
|
return true
|
||||||
|
}
|
||||||
role, ok := callerRole(ctx, teamID)
|
role, ok := callerRole(ctx, teamID)
|
||||||
return ok && role == models.RoleOwner
|
return ok && role == models.RoleOwner
|
||||||
}
|
}
|
||||||
@@ -132,7 +135,7 @@ func handleCreateServiceAccount(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, "initial")
|
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, "initial")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusCreated, map[string]any{"service_account": sa, "key": key})
|
respond(w, http.StatusCreated, map[string]any{"service_account": sa, "key": key})
|
||||||
@@ -233,7 +236,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !callerMayManageServiceAccount(r.Context(), sa) {
|
if !callerMayManageServiceAccount(r.Context(), sa) {
|
||||||
@@ -255,7 +258,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, req.Name)
|
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, req.Name)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusCreated, key)
|
respond(w, http.StatusCreated, key)
|
||||||
@@ -274,7 +277,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !callerMayManageServiceAccount(r.Context(), sa) {
|
if !callerMayManageServiceAccount(r.Context(), sa) {
|
||||||
@@ -290,7 +293,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM service_account_keys WHERE id = $1 AND service_account_id = $2", keyID, sa.ID)
|
"DELETE FROM service_account_keys WHERE id = $1 AND service_account_id = $2", keyID, sa.ID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -325,7 +328,7 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
rows, err := db.QueryContext(r.Context(), query, args...)
|
rows, err := db.QueryContext(r.Context(), query, args...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -335,14 +338,14 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc {
|
|||||||
var sa models.ServiceAccount
|
var sa models.ServiceAccount
|
||||||
var created int64
|
var created int64
|
||||||
if err := rows.Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil {
|
if err := rows.Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
sa.CreatedAt = time.Unix(created, 0).UTC()
|
sa.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
accounts = append(accounts, sa)
|
accounts = append(accounts, sa)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, accounts)
|
respond(w, http.StatusOK, accounts)
|
||||||
|
|||||||
@@ -203,7 +203,7 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
@@ -215,12 +215,12 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc {
|
|||||||
ON CONFLICT (key) DO UPDATE SET
|
ON CONFLICT (key) DO UPDATE SET
|
||||||
value = excluded.value, updated_at = excluded.updated_at`,
|
value = excluded.value, updated_at = excluded.updated_at`,
|
||||||
key, value); err != nil {
|
key, value); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -261,7 +261,7 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
FROM teams t
|
FROM teams t
|
||||||
ORDER BY t.name`)
|
ORDER BY t.name`)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -272,14 +272,14 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
var created int64
|
var created int64
|
||||||
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents,
|
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents,
|
||||||
&t.OIDCMemberGroup, &t.OIDCOwnerGroup); err != nil {
|
&t.OIDCMemberGroup, &t.OIDCOwnerGroup); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
teams = append(teams, t)
|
teams = append(teams, t)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, teams)
|
respond(w, http.StatusOK, teams)
|
||||||
@@ -319,7 +319,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
@@ -333,7 +333,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE m.team_id = $1
|
WHERE m.team_id = $1
|
||||||
ORDER BY u.username`, teamID)
|
ORDER BY u.username`, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -343,14 +343,14 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
var m models.TeamMember
|
var m models.TeamMember
|
||||||
var joined int64
|
var joined int64
|
||||||
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source); err != nil {
|
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
m.JoinedAt = time.Unix(joined, 0).UTC()
|
m.JoinedAt = time.Unix(joined, 0).UTC()
|
||||||
members = append(members, m)
|
members = append(members, m)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -396,7 +396,7 @@ func handleRenameTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -435,7 +435,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
last, err := isLastAdmin(r.Context(), db, id)
|
last, err := isLastAdmin(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last {
|
if last {
|
||||||
@@ -453,7 +453,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
|
|||||||
"UPDATE users SET disabled_at = NULL WHERE id = $1", id)
|
"UPDATE users SET disabled_at = NULL WHERE id = $1", id)
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -475,7 +475,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, id)
|
user, err := fetchUser(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, user)
|
respond(w, http.StatusOK, user)
|
||||||
|
|||||||
@@ -170,13 +170,13 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
|
|||||||
|
|
||||||
hash, err := hashPassword(req.Password)
|
hash, err := hashPassword(req.Password)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
@@ -194,7 +194,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
|
|||||||
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -207,7 +207,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
|
|||||||
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
role = models.RoleOwner
|
role = models.RoleOwner
|
||||||
@@ -216,7 +216,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
|
|||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
||||||
teamID, userID, role); err != nil {
|
teamID, userID, role); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -226,7 +226,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
|
|||||||
res, err := tx.ExecContext(r.Context(),
|
res, err := tx.ExecContext(r.Context(),
|
||||||
"UPDATE invites SET uses = uses + 1 WHERE id = $1 AND uses < max_uses", inv.id)
|
"UPDATE invites SET uses = uses + 1 WHERE id = $1 AND uses < max_uses", inv.id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -236,14 +236,14 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
|
|||||||
}
|
}
|
||||||
|
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
// Signed in immediately: the alternative is a form that says "now go
|
// Signed in immediately: the alternative is a form that says "now go
|
||||||
// and log in", which is the same credential typed twice.
|
// and log in", which is the same credential typed twice.
|
||||||
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, _ := fetchUser(r.Context(), db, userID)
|
user, _ := fetchUser(r.Context(), db, userID)
|
||||||
@@ -291,7 +291,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE team_id = $1
|
WHERE team_id = $1
|
||||||
ORDER BY id DESC`, teamID)
|
ORDER BY id DESC`, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -303,7 +303,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc {
|
|||||||
var revoked *int64
|
var revoked *int64
|
||||||
if err := rows.Scan(&i.ID, &i.TeamID, &i.Role, &created, &expires,
|
if err := rows.Scan(&i.ID, &i.TeamID, &i.Role, &created, &expires,
|
||||||
&i.MaxUses, &i.Uses, &revoked); err != nil {
|
&i.MaxUses, &i.Uses, &revoked); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
i.CreatedAt = time.Unix(created, 0).UTC()
|
i.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
@@ -312,7 +312,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc {
|
|||||||
out = append(out, i)
|
out = append(out, i)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, out)
|
respond(w, http.StatusOK, out)
|
||||||
@@ -356,7 +356,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
|
|
||||||
raw, hash, err := randomToken()
|
raw, hash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// created_by is nullable (ON DELETE SET NULL) for exactly this
|
// created_by is nullable (ON DELETE SET NULL) for exactly this
|
||||||
@@ -382,7 +382,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
RETURNING id, team_id, role, created_at, expires_at, max_uses, uses`,
|
RETURNING id, team_id, role, created_at, expires_at, max_uses, uses`,
|
||||||
hash, teamID, req.Role, createdBy, expires.Unix(), req.MaxUses).
|
hash, teamID, req.Role, createdBy, expires.Unix(), req.MaxUses).
|
||||||
Scan(&out.ID, &out.TeamID, &out.Role, &created, &expiresAt, &out.MaxUses, &out.Uses); err != nil {
|
Scan(&out.ID, &out.TeamID, &out.Role, &created, &expiresAt, &out.MaxUses, &out.Uses); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
out.CreatedAt = time.Unix(created, 0).UTC()
|
out.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
@@ -412,7 +412,7 @@ func handleRevokeInvite(db *sql.DB) http.HandlerFunc {
|
|||||||
"UPDATE invites SET revoked_at = "+nowEpoch+
|
"UPDATE invites SET revoked_at = "+nowEpoch+
|
||||||
" WHERE id = $1 AND team_id = $2 AND revoked_at IS NULL", id, teamID)
|
" WHERE id = $1 AND team_id = $2 AND revoked_at IS NULL", id, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -450,7 +450,7 @@ func handleTestNotification(cfg NotifyConfig, db *sql.DB) http.HandlerFunc {
|
|||||||
var topic *string
|
var topic *string
|
||||||
if err := db.QueryRowContext(r.Context(),
|
if err := db.QueryRowContext(r.Context(),
|
||||||
"SELECT ntfy_topic FROM users WHERE id = $1", caller.ID).Scan(&topic); err != nil {
|
"SELECT ntfy_topic FROM users WHERE id = $1", caller.ID).Scan(&topic); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if topic == nil || *topic == "" {
|
if topic == nil || *topic == "" {
|
||||||
@@ -501,7 +501,7 @@ func handleDismissOnboarding(db *sql.DB) http.HandlerFunc {
|
|||||||
"UPDATE users SET onboarding_dismissed_at = NULL WHERE id = $1", caller.ID)
|
"UPDATE users SET onboarding_dismissed_at = NULL WHERE id = $1", caller.ID)
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ func handleIncidentSimilar(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
out, err := similarIncidents(r.Context(), db, id, limit)
|
out, err := similarIncidents(r.Context(), db, id, limit)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, out)
|
respond(w, http.StatusOK, out)
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ func handleStatsAlerts(db *sql.DB) http.HandlerFunc {
|
|||||||
FROM alerts WHERE %s`, where), args.all()...,
|
FROM alerts WHERE %s`, where), args.all()...,
|
||||||
).Scan(&total, &firing, &resolved)
|
).Scan(&total, &firing, &resolved)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, map[string]int64{
|
respond(w, http.StatusOK, map[string]int64{
|
||||||
@@ -55,7 +55,7 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc {
|
|||||||
ORDER BY cnt DESC
|
ORDER BY cnt DESC
|
||||||
LIMIT %s`, where, args.add(limit)), args.all()...)
|
LIMIT %s`, where, args.add(limit)), args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -68,13 +68,13 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc {
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var e entry
|
var e entry
|
||||||
if err := rows.Scan(&e.Name, &e.Count); err != nil {
|
if err := rows.Scan(&e.Name, &e.Count); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
result = append(result, e)
|
result = append(result, e)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, result)
|
respond(w, http.StatusOK, result)
|
||||||
@@ -93,7 +93,7 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc {
|
|||||||
GROUP BY hr
|
GROUP BY hr
|
||||||
ORDER BY hr ASC`, where), args.all()...)
|
ORDER BY hr ASC`, where), args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -103,13 +103,13 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc {
|
|||||||
var hr int
|
var hr int
|
||||||
var cnt int64
|
var cnt int64
|
||||||
if err := rows.Scan(&hr, &cnt); err != nil {
|
if err := rows.Scan(&hr, &cnt); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
counts[hr] = cnt
|
counts[hr] = cnt
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -129,8 +129,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
|||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
||||||
|
|
||||||
// Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday, the same numbering
|
// Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday.
|
||||||
// SQLite's strftime('%w') returned, so the frontend needs no change.
|
|
||||||
rows, err := db.QueryContext(r.Context(), fmt.Sprintf(`
|
rows, err := db.QueryContext(r.Context(), fmt.Sprintf(`
|
||||||
SELECT EXTRACT(DOW FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS dow,
|
SELECT EXTRACT(DOW FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS dow,
|
||||||
COUNT(*) AS cnt
|
COUNT(*) AS cnt
|
||||||
@@ -139,7 +138,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
|||||||
GROUP BY dow
|
GROUP BY dow
|
||||||
ORDER BY dow ASC`, where), args.all()...)
|
ORDER BY dow ASC`, where), args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -149,13 +148,13 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
|||||||
var dow int
|
var dow int
|
||||||
var cnt int64
|
var cnt int64
|
||||||
if err := rows.Scan(&dow, &cnt); err != nil {
|
if err := rows.Scan(&dow, &cnt); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
counts[dow] = cnt
|
counts[dow] = cnt
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -198,7 +197,7 @@ func handleStatsIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
FROM incidents WHERE %s`, where), args.all()...,
|
FROM incidents WHERE %s`, where), args.all()...,
|
||||||
).Scan(&total, &triggered, &acknowledged, &resolved, &mtta, &mttr)
|
).Scan(&total, &triggered, &acknowledged, &resolved, &mtta, &mttr)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -13,19 +13,12 @@ import (
|
|||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
)
|
)
|
||||||
|
|
||||||
// handleListTeams lists the caller's own teams, each with their role in it,
|
// handleListTeams lists the caller's own teams, each with their role in it.
|
||||||
// or — with ?name= — looks up one team by exact name regardless of caller
|
// An administrator listing every team goes through the admin endpoint instead:
|
||||||
// identity (TEAM-LOOKUP.md). An administrator listing every team goes
|
// this answers "what am I part of", which is what the UI's team filter and the
|
||||||
// through the admin endpoint instead: the no-name case here answers "what am
|
// combined queue are built from.
|
||||||
// I part of", which is what the UI's team filter and the combined queue are
|
|
||||||
// built from.
|
|
||||||
func handleListTeams(db *sql.DB) http.HandlerFunc {
|
func handleListTeams(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
if name := strings.TrimSpace(r.URL.Query().Get("name")); name != "" {
|
|
||||||
handleListTeamsByName(db, w, r, name)
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
caller, _ := userFromContext(r.Context())
|
caller, _ := userFromContext(r.Context())
|
||||||
rows, err := db.QueryContext(r.Context(), `
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
SELECT t.id, t.name, t.created_at, m.role, m.source
|
SELECT t.id, t.name, t.created_at, m.role, m.source
|
||||||
@@ -34,7 +27,7 @@ func handleListTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE m.user_id = $1
|
WHERE m.user_id = $1
|
||||||
ORDER BY t.name`, caller.ID)
|
ORDER BY t.name`, caller.ID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -44,45 +37,20 @@ func handleListTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
var t models.Team
|
var t models.Team
|
||||||
var created int64
|
var created int64
|
||||||
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
|
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
teams = append(teams, t)
|
teams = append(teams, t)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, teams)
|
respond(w, http.StatusOK, teams)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// handleListTeamsByName answers "is there a team named exactly this", open to
|
|
||||||
// any authenticated caller including a service account (TEAM-LOOKUP.md) —
|
|
||||||
// mirrors handleListServiceAccounts' own ?name= lookup: a one-or-zero-length
|
|
||||||
// array, never an error on no match, and no caller-identity filtering at
|
|
||||||
// all, since what it discloses (a name is taken, nothing about who's in it
|
|
||||||
// or any of its data) is the same low sensitivity that lookup already
|
|
||||||
// accepts for service-account names.
|
|
||||||
func handleListTeamsByName(db *sql.DB, w http.ResponseWriter, r *http.Request, name string) {
|
|
||||||
var t models.Team
|
|
||||||
var created int64
|
|
||||||
err := db.QueryRowContext(r.Context(),
|
|
||||||
"SELECT id, name, created_at FROM teams WHERE name = $1", name,
|
|
||||||
).Scan(&t.ID, &t.Name, &created)
|
|
||||||
if errors.Is(err, sql.ErrNoRows) {
|
|
||||||
respond(w, http.StatusOK, []models.Team{})
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if err != nil {
|
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
|
||||||
return
|
|
||||||
}
|
|
||||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
|
||||||
respond(w, http.StatusOK, []models.Team{t})
|
|
||||||
}
|
|
||||||
|
|
||||||
// handleUserTeams lists one user's teams, for the admin page's per-user view:
|
// handleUserTeams lists one user's teams, for the admin page's per-user view:
|
||||||
// "what is this person in", which /api/teams cannot answer because it is always
|
// "what is this person in", which /api/teams cannot answer because it is always
|
||||||
// about the caller.
|
// about the caller.
|
||||||
@@ -106,7 +74,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
var exists bool
|
var exists bool
|
||||||
if err := db.QueryRowContext(r.Context(),
|
if err := db.QueryRowContext(r.Context(),
|
||||||
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1)", id).Scan(&exists); err != nil {
|
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1)", id).Scan(&exists); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !exists {
|
if !exists {
|
||||||
@@ -121,7 +89,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE m.user_id = $1
|
WHERE m.user_id = $1
|
||||||
ORDER BY t.name`, id)
|
ORDER BY t.name`, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -131,14 +99,14 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
var t models.Team
|
var t models.Team
|
||||||
var created int64
|
var created int64
|
||||||
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
|
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
teams = append(teams, t)
|
teams = append(teams, t)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, teams)
|
respond(w, http.StatusOK, teams)
|
||||||
@@ -161,6 +129,12 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
var req struct {
|
var req struct {
|
||||||
Name string `json:"name"`
|
Name string `json:"name"`
|
||||||
|
// ExternalID makes the call idempotent for automation: a team
|
||||||
|
// already carrying it is returned as-is (200) instead of created, so
|
||||||
|
// a client that crashed between the POST and recording the id finds
|
||||||
|
// its own team again. Instance-scoped service accounts only; a name
|
||||||
|
// that belongs to a different team is still a 409.
|
||||||
|
ExternalID string `json:"external_id"`
|
||||||
}
|
}
|
||||||
if err := decodeJSON(r, &req); err != nil {
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
@@ -173,6 +147,10 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
caller, isUser := userFromContext(r.Context())
|
caller, isUser := userFromContext(r.Context())
|
||||||
|
if req.ExternalID != "" && !isInstanceServiceAccount(r.Context()) {
|
||||||
|
respond(w, http.StatusForbidden, errResp("external_id is for instance-scoped service accounts"))
|
||||||
|
return
|
||||||
|
}
|
||||||
if !isUser && !isInstanceServiceAccount(r.Context()) {
|
if !isUser && !isInstanceServiceAccount(r.Context()) {
|
||||||
// A team-scoped service account authenticates as owner of exactly
|
// A team-scoped service account authenticates as owner of exactly
|
||||||
// one team already (see serveAsServiceAccount); letting it create
|
// one team already (see serveAsServiceAccount); letting it create
|
||||||
@@ -183,37 +161,57 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
var team models.Team
|
var team models.Team
|
||||||
var created int64
|
var created int64
|
||||||
|
if req.ExternalID != "" {
|
||||||
|
err := tx.QueryRowContext(r.Context(),
|
||||||
|
"SELECT id, name, created_at FROM teams WHERE external_id = $1", req.ExternalID).
|
||||||
|
Scan(&team.ID, &team.Name, &created)
|
||||||
|
if err == nil {
|
||||||
|
team.ExternalID = &req.ExternalID
|
||||||
|
team.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
respond(w, http.StatusOK, team)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !errors.Is(err, sql.ErrNoRows) {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var externalID *string
|
||||||
|
if req.ExternalID != "" {
|
||||||
|
externalID = &req.ExternalID
|
||||||
|
}
|
||||||
if err := tx.QueryRowContext(r.Context(),
|
if err := tx.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO teams (name) VALUES ($1) RETURNING id, name, created_at",
|
"INSERT INTO teams (name, external_id) VALUES ($1, $2) RETURNING id, name, created_at",
|
||||||
req.Name).Scan(&team.ID, &team.Name, &created); err != nil {
|
req.Name, externalID).Scan(&team.ID, &team.Name, &created); err != nil {
|
||||||
if isUniqueViolation(err) {
|
if isUniqueViolation(err) {
|
||||||
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if isUser {
|
if isUser {
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
||||||
team.ID, caller.ID, models.RoleOwner); err != nil {
|
team.ID, caller.ID, models.RoleOwner); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
team.CreatedAt = time.Unix(created, 0).UTC()
|
team.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
team.ExternalID = externalID
|
||||||
if isUser {
|
if isUser {
|
||||||
team.Role = models.RoleOwner
|
team.Role = models.RoleOwner
|
||||||
}
|
}
|
||||||
@@ -241,7 +239,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
if err := db.QueryRowContext(r.Context(),
|
if err := db.QueryRowContext(r.Context(),
|
||||||
"SELECT COUNT(*) FROM incidents WHERE team_id = $1 AND resolved_at IS NULL", teamID).
|
"SELECT COUNT(*) FROM incidents WHERE team_id = $1 AND resolved_at IS NULL", teamID).
|
||||||
Scan(&open); err != nil {
|
Scan(&open); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if open > 0 {
|
if open > 0 {
|
||||||
@@ -251,7 +249,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
res, err := db.ExecContext(r.Context(), "DELETE FROM teams WHERE id = $1", teamID)
|
res, err := db.ExecContext(r.Context(), "DELETE FROM teams WHERE id = $1", teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -322,7 +320,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE m.team_id = $1
|
WHERE m.team_id = $1
|
||||||
ORDER BY u.username`, teamID, todayUTC())
|
ORDER BY u.username`, teamID, todayUTC())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -334,7 +332,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
|
|||||||
var hasTopic, disabled bool
|
var hasTopic, disabled bool
|
||||||
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source,
|
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source,
|
||||||
&hasTopic, &disabled, &lastActive, &m.OnCall, &m.NextShift); err != nil {
|
&hasTopic, &disabled, &lastActive, &m.OnCall, &m.NextShift); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
m.JoinedAt = time.Unix(joined, 0).UTC()
|
m.JoinedAt = time.Unix(joined, 0).UTC()
|
||||||
@@ -360,7 +358,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
|
|||||||
members = append(members, m)
|
members = append(members, m)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, members)
|
respond(w, http.StatusOK, members)
|
||||||
@@ -396,7 +394,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if managed, err := isSSOManagedMember(r.Context(), db, teamID, req.UserID); err != nil {
|
if managed, err := isSSOManagedMember(r.Context(), db, teamID, req.UserID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
} else if managed {
|
} else if managed {
|
||||||
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
|
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
|
||||||
@@ -408,7 +406,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
if req.Role == models.RoleMember {
|
if req.Role == models.RoleMember {
|
||||||
last, err := isLastTeamOwner(r.Context(), db, teamID, req.UserID)
|
last, err := isLastTeamOwner(r.Context(), db, teamID, req.UserID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last {
|
if last {
|
||||||
@@ -453,7 +451,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if managed, err := isSSOManagedMember(r.Context(), db, teamID, userID); err != nil {
|
if managed, err := isSSOManagedMember(r.Context(), db, teamID, userID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
} else if managed {
|
} else if managed {
|
||||||
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
|
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
|
||||||
@@ -462,7 +460,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
last, err := isLastTeamOwner(r.Context(), db, teamID, userID)
|
last, err := isLastTeamOwner(r.Context(), db, teamID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last {
|
if last {
|
||||||
@@ -473,7 +471,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM team_members WHERE team_id = $1 AND user_id = $2", teamID, userID)
|
"DELETE FROM team_members WHERE team_id = $1 AND user_id = $2", teamID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -572,7 +570,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE i.team_id = $1
|
WHERE i.team_id = $1
|
||||||
ORDER BY i.id`, teamID, now.Add(-sourceQuietAfter).Unix())
|
ORDER BY i.id`, teamID, now.Add(-sourceQuietAfter).Unix())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -584,7 +582,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc {
|
|||||||
var lastUsed, lastAlert *int64
|
var lastUsed, lastAlert *int64
|
||||||
if err := rows.Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created, &lastUsed,
|
if err := rows.Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created, &lastUsed,
|
||||||
&lastAlert, &i.Alerts24h); err != nil {
|
&lastAlert, &i.Alerts24h); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
i.CreatedAt = time.Unix(created, 0).UTC()
|
i.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
@@ -601,7 +599,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc {
|
|||||||
integrations = append(integrations, i)
|
integrations = append(integrations, i)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, integrations)
|
respond(w, http.StatusOK, integrations)
|
||||||
@@ -644,7 +642,7 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
|
|
||||||
raw, hash, err := randomToken()
|
raw, hash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -656,7 +654,11 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
RETURNING id, team_id, kind, name, created_at`,
|
RETURNING id, team_id, kind, name, created_at`,
|
||||||
teamID, req.Kind, req.Name, hash).
|
teamID, req.Kind, req.Name, hash).
|
||||||
Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created); err != nil {
|
Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
i.CreatedAt = time.Unix(created, 0).UTC()
|
i.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
@@ -703,7 +705,11 @@ func handleRenameIntegration(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE integrations SET name = $1 WHERE id = $2 AND team_id = $3", req.Name, id, teamID)
|
"UPDATE integrations SET name = $1 WHERE id = $2 AND team_id = $3", req.Name, id, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -732,7 +738,7 @@ func handleDeleteIntegration(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM integrations WHERE id = $1 AND team_id = $2", id, teamID)
|
"DELETE FROM integrations WHERE id = $1 AND team_id = $2", id, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -790,16 +796,6 @@ func teamParam(w http.ResponseWriter, r *http.Request) (int64, bool) {
|
|||||||
return id, true
|
return id, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// defaultTeamID is the oldest team, which on an upgraded install is the
|
|
||||||
// "Default" team every pre-teams row was moved into and on a fresh one is the
|
|
||||||
// team migration 003 creates. Bootstrap puts the first user in it, so somebody
|
|
||||||
// signing in to a new server lands somewhere rather than in no team at all.
|
|
||||||
func defaultTeamID(ctx context.Context, db *sql.DB) (int64, error) {
|
|
||||||
var id int64
|
|
||||||
err := db.QueryRowContext(ctx, "SELECT id FROM teams ORDER BY id LIMIT 1").Scan(&id)
|
|
||||||
return id, err
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// A team's dead man's switches
|
// A team's dead man's switches
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -833,12 +829,12 @@ func handleListTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
set, err := deadmanSetForTeam(r.Context(), db, teamID)
|
set, err := deadmanSetForTeam(r.Context(), db, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
out, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
|
out, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, out)
|
respond(w, http.StatusOK, out)
|
||||||
@@ -901,7 +897,11 @@ func handleCreateTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
||||||
VALUES ($1, $2, $3, $4, $5) RETURNING id`,
|
VALUES ($1, $2, $3, $4, $5) RETURNING id`,
|
||||||
teamID, req.Name, m.config(), req.TimeoutSeconds, req.Severity).Scan(&id); err != nil {
|
teamID, req.Name, m.config(), req.TimeoutSeconds, req.Severity).Scan(&id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -975,7 +975,11 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE id = $5 AND team_id = $6`,
|
WHERE id = $5 AND team_id = $6`,
|
||||||
req.Name, m.config(), req.TimeoutSeconds, req.Severity, switchID, teamID)
|
req.Name, m.config(), req.TimeoutSeconds, req.Severity, switchID, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -990,12 +994,12 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
// handleListTeamDeadman would give it.
|
// handleListTeamDeadman would give it.
|
||||||
set, err := deadmanSetForTeam(r.Context(), db, teamID)
|
set, err := deadmanSetForTeam(r.Context(), db, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
statuses, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
|
statuses, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
for _, s := range statuses {
|
for _, s := range statuses {
|
||||||
@@ -1004,7 +1008,7 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1029,7 +1033,7 @@ func handleDeleteTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM deadman_switches WHERE id = $1 AND team_id = $2", switchID, teamID)
|
"DELETE FROM deadman_switches WHERE id = $1 AND team_id = $2", switchID, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
|||||||
@@ -6,8 +6,6 @@ import (
|
|||||||
"io"
|
"io"
|
||||||
"net/http"
|
"net/http"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// The whole point of #4: two teams sharing one server must not see each other's
|
// The whole point of #4: two teams sharing one server must not see each other's
|
||||||
@@ -146,7 +144,6 @@ func TestTeams_IncidentsAreScopedToTheReceivingTeam(t *testing.T) {
|
|||||||
otherID := int64(blueIncidents[0]["id"].(float64))
|
otherID := int64(blueIncidents[0]["id"].(float64))
|
||||||
for _, path := range []string{
|
for _, path := range []string{
|
||||||
"/api/incidents/" + id64(otherID),
|
"/api/incidents/" + id64(otherID),
|
||||||
"/api/incidents/" + id64(otherID) + "/alerts",
|
|
||||||
"/api/incidents/" + id64(otherID) + "/timeline",
|
"/api/incidents/" + id64(otherID) + "/timeline",
|
||||||
} {
|
} {
|
||||||
resp := red.call(http.MethodGet, path, nil)
|
resp := red.call(http.MethodGet, path, nil)
|
||||||
@@ -461,70 +458,58 @@ func TestTeams_OutsiderSeesNothing(t *testing.T) {
|
|||||||
// GET /api/teams?name= (TEAM-LOOKUP.md)
|
// GET /api/teams?name= (TEAM-LOOKUP.md)
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
func TestListTeamsByName_FindsExactMatch(t *testing.T) {
|
// Everybody signed in can list users to name them, but only an admin (or the
|
||||||
s := newTS(t)
|
// row's owner) sees an email or an ntfy topic, which is a publish secret.
|
||||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
func TestListUsers_RedactsEmailAndTopicForOthers(t *testing.T) {
|
||||||
teamID := createTeamAs(t, s, instanceKey, "platform")
|
|
||||||
|
|
||||||
teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=platform", nil))
|
|
||||||
if len(teams) != 1 {
|
|
||||||
t.Fatalf("expected exactly one match for ?name=platform, got %d: %v", len(teams), teams)
|
|
||||||
}
|
|
||||||
if int64(teams[0]["id"].(float64)) != teamID {
|
|
||||||
t.Errorf("id = %v, want %d", teams[0]["id"], teamID)
|
|
||||||
}
|
|
||||||
// No membership, so no role to report (models.Team's own doc comment:
|
|
||||||
// "empty when nobody in particular is asking").
|
|
||||||
if _, has := teams[0]["role"]; has {
|
|
||||||
t.Errorf("expected no role on a name-lookup match, got %v", teams[0]["role"])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestListTeamsByName_NoMatchIsAnEmptyArrayNotAnError(t *testing.T) {
|
|
||||||
s := newTS(t)
|
|
||||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
|
||||||
|
|
||||||
resp := s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=does-not-exist", nil)
|
|
||||||
if resp.StatusCode != http.StatusOK {
|
|
||||||
t.Fatalf("expected 200 on no match, got %d", resp.StatusCode)
|
|
||||||
}
|
|
||||||
teams := list(t, resp)
|
|
||||||
if len(teams) != 0 {
|
|
||||||
t.Errorf("expected an empty array, got %v", teams)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// The actual motivating scenario (TEAM-LOOKUP.md): a service account that
|
|
||||||
// already created a team, interrupted before it could remember the id,
|
|
||||||
// recovers it via ?name= on the same name its own POST 409s on.
|
|
||||||
func TestListTeamsByName_RecoversAfterCreateConflict(t *testing.T) {
|
|
||||||
s := newTS(t)
|
|
||||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
|
||||||
original := createTeamAs(t, s, instanceKey, "recovered")
|
|
||||||
|
|
||||||
conflict := s.reqAs(t, instanceKey, http.MethodPost, "/api/teams", map[string]string{"name": "recovered"})
|
|
||||||
if conflict.StatusCode != http.StatusConflict {
|
|
||||||
t.Fatalf("expected 409 recreating the same name, got %d", conflict.StatusCode)
|
|
||||||
}
|
|
||||||
conflict.Body.Close()
|
|
||||||
|
|
||||||
teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=recovered", nil))
|
|
||||||
if len(teams) != 1 || int64(teams[0]["id"].(float64)) != original {
|
|
||||||
t.Fatalf("expected to recover the original team %d via ?name=, got %v", original, teams)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Not gated by isInstanceServiceAccount or AdminOnly (TEAM-LOOKUP.md): any
|
|
||||||
// authenticated caller may ask whether a name is taken, the same low
|
|
||||||
// sensitivity GET /api/service-accounts?name= already accepts.
|
|
||||||
func TestListTeamsByName_OpenToAnyAuthenticatedCaller(t *testing.T) {
|
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
red := newTeam(t, s, "red")
|
red := newTeam(t, s, "red")
|
||||||
_ = createTeamAs(t, s, s.key, "blue-target")
|
_ = newTeam(t, s, "blue")
|
||||||
|
s.exec(t, "UPDATE users SET ntfy_topic = 'secret-topic'")
|
||||||
|
|
||||||
// red's own member, not a member of "blue-target", still gets a match.
|
var asAdmin []map[string]any
|
||||||
teams := list(t, red.call(http.MethodGet, "/api/teams?name=blue-target", nil))
|
decode(t, s.req(t, http.MethodGet, "/api/users", nil), &asAdmin)
|
||||||
if len(teams) != 1 || teams[0]["name"] != "blue-target" {
|
for _, u := range asAdmin {
|
||||||
t.Errorf("expected a non-member caller to still find the team by name, got %v", teams)
|
if u["email"] == "" || u["ntfy_topic"] != "secret-topic" {
|
||||||
|
t.Errorf("admin should see everything, got %v", u)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
asMember := list(t, red.call(http.MethodGet, "/api/users", nil))
|
||||||
|
if len(asMember) < 3 {
|
||||||
|
t.Fatalf("expected the whole user list, got %v", asMember)
|
||||||
|
}
|
||||||
|
for _, u := range asMember {
|
||||||
|
own := u["username"] == "red-user"
|
||||||
|
if own != (u["email"] != "") || own != (u["ntfy_topic"] != nil) {
|
||||||
|
t.Errorf("only red-user's own row should keep email and topic, got %v", u)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Names identify integrations and switches within a team.
|
||||||
|
func TestTeamNames_AreUniquePerTeam(t *testing.T) {
|
||||||
|
s := newTS(t) // creates one integration named "test" in the default team
|
||||||
|
|
||||||
|
dup := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/integrations", map[string]string{"name": "test"})
|
||||||
|
dup.Body.Close()
|
||||||
|
if dup.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("duplicate integration name: expected 409, got %d", dup.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
body := map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 60, "severity": "critical"}
|
||||||
|
first := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
|
||||||
|
first.Body.Close()
|
||||||
|
second := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
|
||||||
|
second.Body.Close()
|
||||||
|
if first.StatusCode != http.StatusCreated || second.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("duplicate switch name: expected 201 then 409, got %d then %d", first.StatusCode, second.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The same name in another team is fine.
|
||||||
|
other := newTeam(t, s, "elsewhere")
|
||||||
|
ok := s.req(t, http.MethodPost, "/api/teams/"+id64(other.id)+"/integrations", map[string]string{"name": "test"})
|
||||||
|
ok.Body.Close()
|
||||||
|
if ok.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("same name in another team: expected 201, got %d", ok.StatusCode)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -13,9 +13,8 @@ import (
|
|||||||
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Tests run against a real Postgres, because the server does. SQLite's
|
// Tests run against a real Postgres, because the server does. Isolation is
|
||||||
// ":memory:" gave every test a private database for free; Postgres has no
|
// bought with a schema per test.
|
||||||
// equivalent, so isolation is bought with a schema per test.
|
|
||||||
//
|
//
|
||||||
// A schema rather than a database: CREATE DATABASE copies a template on disk and
|
// A schema rather than a database: CREATE DATABASE copies a template on disk and
|
||||||
// costs a hundred milliseconds or so each time, while CREATE SCHEMA plus the one
|
// costs a hundred milliseconds or so each time, while CREATE SCHEMA plus the one
|
||||||
|
|||||||
@@ -15,6 +15,10 @@ import (
|
|||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// bootstrapLockKey is the transaction-scoped advisory lock handleBootstrap
|
||||||
|
// holds; distinct from the migration and notifier keys.
|
||||||
|
const bootstrapLockKey = 0x7465726475744254
|
||||||
|
|
||||||
func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
var req struct {
|
var req struct {
|
||||||
@@ -40,15 +44,30 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
h, err := hashPassword(req.Password)
|
h, err := hashPassword(req.Password)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
passwordHash = &h
|
passwordHash = &h
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Check-then-insert has to be one atomic step: two concurrent calls on
|
||||||
|
// an empty install would otherwise both see zero users and both create
|
||||||
|
// an admin. The transaction-scoped lock serialises them, and the loser
|
||||||
|
// sees the winner's row.
|
||||||
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
if _, err := tx.ExecContext(r.Context(), "SELECT pg_advisory_xact_lock($1)", bootstrapLockKey); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
var count int
|
var count int
|
||||||
if err := db.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil {
|
if err := tx.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if count > 0 {
|
if count > 0 {
|
||||||
@@ -57,34 +76,29 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
var userID int64
|
var userID int64
|
||||||
if err := db.QueryRowContext(r.Context(),
|
if err := tx.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id",
|
"INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id",
|
||||||
req.Username, req.Email, passwordHash).Scan(&userID); err != nil {
|
req.Username, req.Email, passwordHash).Scan(&userID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
raw, hash, err := randomToken()
|
raw, hash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
var keyID int64
|
var keyID int64
|
||||||
if err := db.QueryRowContext(r.Context(),
|
if err := tx.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id",
|
"INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id",
|
||||||
userID, hash, "bootstrap").Scan(&keyID); err != nil {
|
userID, hash, "bootstrap").Scan(&keyID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
// The default team exists from migration 003, on a fresh install too.
|
if err := tx.Commit(); err != nil {
|
||||||
// Without a membership the first user signs in to a working server with
|
serverError(w, r, err)
|
||||||
// no queue, no schedule and nowhere for an integration to hang off.
|
return
|
||||||
if teamID, err := defaultTeamID(r.Context(), db); err == nil {
|
|
||||||
db.ExecContext(r.Context(), //nolint:errcheck
|
|
||||||
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3) "+
|
|
||||||
"ON CONFLICT (team_id, user_id) DO NOTHING",
|
|
||||||
teamID, userID, models.RoleOwner)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
user, _ := fetchUser(r.Context(), db, userID)
|
user, _ := fetchUser(r.Context(), db, userID)
|
||||||
@@ -93,12 +107,19 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handleListUsers is readable by anyone signed in, because the assignment
|
||||||
|
// control and the schedule need to name people. What it returns about other
|
||||||
|
// people is therefore only what naming them takes: email and ntfy_topic are
|
||||||
|
// blanked unless the caller is an admin or the row is their own. The topic in
|
||||||
|
// particular is a publish secret.
|
||||||
func handleListUsers(db *sql.DB) http.HandlerFunc {
|
func handleListUsers(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
caller, _ := userFromContext(r.Context())
|
||||||
|
seeAll := caller.IsAdmin
|
||||||
rows, err := db.QueryContext(r.Context(),
|
rows, err := db.QueryContext(r.Context(),
|
||||||
"SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users ORDER BY id")
|
"SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users ORDER BY id")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -109,15 +130,19 @@ func handleListUsers(db *sql.DB) http.HandlerFunc {
|
|||||||
var ts int64
|
var ts int64
|
||||||
var disabled *int64
|
var disabled *int64
|
||||||
if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled); err != nil {
|
if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
u.CreatedAt = time.Unix(ts, 0).UTC()
|
u.CreatedAt = time.Unix(ts, 0).UTC()
|
||||||
u.DisabledAt = unixPtr(disabled)
|
u.DisabledAt = unixPtr(disabled)
|
||||||
|
if !seeAll && u.ID != caller.ID {
|
||||||
|
u.Email = ""
|
||||||
|
u.NtfyTopic = nil
|
||||||
|
}
|
||||||
users = append(users, u)
|
users = append(users, u)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, users)
|
respond(w, http.StatusOK, users)
|
||||||
@@ -147,7 +172,7 @@ func handleCreateUser(db *sql.DB) http.HandlerFunc {
|
|||||||
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, _ := fetchUser(r.Context(), db, id)
|
user, _ := fetchUser(r.Context(), db, id)
|
||||||
@@ -185,7 +210,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE users SET ntfy_topic = $1 WHERE id = $2", topic, id)
|
"UPDATE users SET ntfy_topic = $1 WHERE id = $2", topic, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -195,7 +220,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, id)
|
user, err := fetchUser(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, user)
|
respond(w, http.StatusOK, user)
|
||||||
@@ -217,7 +242,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
} else if last {
|
} else if last {
|
||||||
respond(w, http.StatusConflict, errResp("cannot delete the last administrator"))
|
respond(w, http.StatusConflict, errResp("cannot delete the last administrator"))
|
||||||
@@ -226,7 +251,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id)
|
res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
n, _ := res.RowsAffected()
|
n, _ := res.RowsAffected()
|
||||||
@@ -283,7 +308,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
raw, hash, err := randomToken()
|
raw, hash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
var expiresAt *int64
|
var expiresAt *int64
|
||||||
@@ -298,7 +323,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
|||||||
if err := db.QueryRowContext(r.Context(),
|
if err := db.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO api_keys (user_id, key_hash, name, expires_at) VALUES ($1, $2, $3, $4) RETURNING id",
|
"INSERT INTO api_keys (user_id, key_hash, name, expires_at) VALUES ($1, $2, $3, $4) RETURNING id",
|
||||||
userID, hash, req.Name, expiresAt).Scan(&keyID); err != nil {
|
userID, hash, req.Name, expiresAt).Scan(&keyID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
key := models.APIKey{
|
key := models.APIKey{
|
||||||
@@ -329,7 +354,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc {
|
|||||||
`SELECT id, name, created_at, last_used_at, expires_at
|
`SELECT id, name, created_at, last_used_at, expires_at
|
||||||
FROM api_keys WHERE user_id = $1 ORDER BY created_at DESC`, userID)
|
FROM api_keys WHERE user_id = $1 ORDER BY created_at DESC`, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -340,7 +365,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc {
|
|||||||
var created int64
|
var created int64
|
||||||
var lastUsed, expires *int64
|
var lastUsed, expires *int64
|
||||||
if err := rows.Scan(&k.ID, &k.Name, &created, &lastUsed, &expires); err != nil {
|
if err := rows.Scan(&k.ID, &k.Name, &created, &lastUsed, &expires); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
k.UserID = userID
|
k.UserID = userID
|
||||||
@@ -350,7 +375,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc {
|
|||||||
keys = append(keys, k)
|
keys = append(keys, k)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, keys)
|
respond(w, http.StatusOK, keys)
|
||||||
@@ -376,7 +401,7 @@ func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM api_keys WHERE id = $1 AND user_id = $2", keyID, userID)
|
"DELETE FROM api_keys WHERE id = $1 AND user_id = $2", keyID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
n, _ := res.RowsAffected()
|
n, _ := res.RowsAffected()
|
||||||
@@ -442,7 +467,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
|||||||
if err := db.QueryRowContext(r.Context(),
|
if err := db.QueryRowContext(r.Context(),
|
||||||
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin AND admin_source = 'oidc')",
|
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin AND admin_source = 'oidc')",
|
||||||
id).Scan(&managed); err != nil {
|
id).Scan(&managed); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if managed {
|
if managed {
|
||||||
@@ -456,7 +481,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
} else if last {
|
} else if last {
|
||||||
respond(w, http.StatusConflict, errResp("cannot revoke the last administrator"))
|
respond(w, http.StatusConflict, errResp("cannot revoke the last administrator"))
|
||||||
@@ -467,7 +492,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id)
|
"UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -477,7 +502,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, id)
|
user, err := fetchUser(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, user)
|
respond(w, http.StatusOK, user)
|
||||||
|
|||||||
@@ -5,17 +5,21 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"net/url"
|
"net/url"
|
||||||
"os"
|
"os"
|
||||||
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// MinOperatorKeyLength is the shortest TERDUT_OPERATOR_KEY accepted: it is a
|
||||||
|
// bearer credential with instance reach, so a short one is refused outright.
|
||||||
|
const MinOperatorKeyLength = 32
|
||||||
|
|
||||||
type Config struct {
|
type Config struct {
|
||||||
Addr string
|
Addr string
|
||||||
|
|
||||||
// DSN is the Postgres connection string, e.g.
|
// DSN is the Postgres connection string, e.g.
|
||||||
// postgres://terdut:secret@host:5432/terdut?sslmode=require. Required:
|
// postgres://terdut:secret@host:5432/terdut?sslmode=require. Required:
|
||||||
// unlike the SQLite path it replaced there is no sensible default, and a
|
// there is no sensible default, and a server that silently came up against the wrong database would be worse
|
||||||
// server that silently came up against the wrong database would be worse
|
|
||||||
// than one that refuses to start.
|
// than one that refuses to start.
|
||||||
DSN string
|
DSN string
|
||||||
|
|
||||||
@@ -26,25 +30,6 @@ type Config struct {
|
|||||||
// repeat_interval (default 4h), which is what refreshes the alert.
|
// repeat_interval (default 4h), which is what refreshes the alert.
|
||||||
StaleAfter time.Duration
|
StaleAfter time.Duration
|
||||||
|
|
||||||
// DeadmanMatchers selects the alerts that are heartbeats rather than
|
|
||||||
// problems: receiving one opens no incident, and the absence of one does.
|
|
||||||
//
|
|
||||||
// ";" separates matchers, "," the label conditions within one, "=" is exact
|
|
||||||
// equality — `alertname=Watchdog,cluster=prod; alertname=Heartbeat`. Every
|
|
||||||
// matcher must name an alertname. See api.ParseDeadmanConfig.
|
|
||||||
DeadmanMatchers string
|
|
||||||
|
|
||||||
// DeadmanTimeout is how long a heartbeat may go unheard before its switch is
|
|
||||||
// declared dead. It must be *shorter* than the Alertmanager repeat_interval
|
|
||||||
// of the route carrying the heartbeat — the opposite of StaleAfter, and the
|
|
||||||
// reason a dead man's switch usually wants a route of its own. Zero disables
|
|
||||||
// dead man's switch handling entirely.
|
|
||||||
DeadmanTimeout time.Duration
|
|
||||||
|
|
||||||
// DeadmanSeverity is the severity a dead man's switch incident opens at.
|
|
||||||
// These incidents have no member alerts to derive one from.
|
|
||||||
DeadmanSeverity string
|
|
||||||
|
|
||||||
// NtfyURL is the ntfy server push notifications are published to. Empty
|
// NtfyURL is the ntfy server push notifications are published to. Empty
|
||||||
// disables notifications entirely.
|
// disables notifications entirely.
|
||||||
NtfyURL string
|
NtfyURL string
|
||||||
@@ -70,6 +55,20 @@ type Config struct {
|
|||||||
// Config, which is what a test or a new caller builds, keeps passwords working.
|
// Config, which is what a test or a new caller builds, keeps passwords working.
|
||||||
DisablePasswordLogin bool
|
DisablePasswordLogin bool
|
||||||
|
|
||||||
|
// OperatorKey, when set, is the credential of the instance-scoped service
|
||||||
|
// account "terdut-operator", created or re-keyed at every start. It is how
|
||||||
|
// terdut-operator gets in without a bootstrap handshake: the operator
|
||||||
|
// generates the key, hands it to the server here, and uses it as its bearer
|
||||||
|
// token. Empty means no such account is managed.
|
||||||
|
OperatorKey string
|
||||||
|
|
||||||
|
// TrustedProxies is how many reverse proxies sit in front of the server and
|
||||||
|
// append to X-Forwarded-For. The per-address rate limits take the client
|
||||||
|
// address that many entries from the right, because everything further left
|
||||||
|
// is whatever the client chose to send. 0 ignores the header and uses the
|
||||||
|
// connection's own address.
|
||||||
|
TrustedProxies int
|
||||||
|
|
||||||
// OIDC configures single sign-on. The zero value, with no Issuer, is off.
|
// OIDC configures single sign-on. The zero value, with no Issuer, is off.
|
||||||
OIDC OIDC
|
OIDC OIDC
|
||||||
|
|
||||||
@@ -133,24 +132,12 @@ func Load() Config {
|
|||||||
if addr == "" {
|
if addr == "" {
|
||||||
addr = ":8080"
|
addr = ":8080"
|
||||||
}
|
}
|
||||||
deadmanMatchers := os.Getenv("TERDUT_DEADMAN_MATCHERS")
|
|
||||||
if deadmanMatchers == "" {
|
|
||||||
deadmanMatchers = "alertname=Watchdog"
|
|
||||||
}
|
|
||||||
deadmanSeverity := os.Getenv("TERDUT_DEADMAN_SEVERITY")
|
|
||||||
if deadmanSeverity == "" {
|
|
||||||
deadmanSeverity = "critical"
|
|
||||||
}
|
|
||||||
return Config{
|
return Config{
|
||||||
Addr: addr,
|
Addr: addr,
|
||||||
DSN: os.Getenv("TERDUT_DB_DSN"),
|
DSN: os.Getenv("TERDUT_DB_DSN"),
|
||||||
ArchiveAfter: duration("TERDUT_ARCHIVE_AFTER", 7*24*time.Hour),
|
ArchiveAfter: duration("TERDUT_ARCHIVE_AFTER", 7*24*time.Hour),
|
||||||
StaleAfter: duration("TERDUT_STALE_AFTER", 6*time.Hour),
|
StaleAfter: duration("TERDUT_STALE_AFTER", 6*time.Hour),
|
||||||
|
|
||||||
DeadmanMatchers: deadmanMatchers,
|
|
||||||
DeadmanTimeout: duration("TERDUT_DEADMAN_TIMEOUT", 15*time.Minute),
|
|
||||||
DeadmanSeverity: deadmanSeverity,
|
|
||||||
|
|
||||||
NtfyURL: os.Getenv("TERDUT_NTFY_URL"),
|
NtfyURL: os.Getenv("TERDUT_NTFY_URL"),
|
||||||
NtfyToken: os.Getenv("TERDUT_NTFY_TOKEN"),
|
NtfyToken: os.Getenv("TERDUT_NTFY_TOKEN"),
|
||||||
NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"),
|
NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"),
|
||||||
@@ -158,7 +145,10 @@ func Load() Config {
|
|||||||
NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute),
|
NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute),
|
||||||
|
|
||||||
DisablePasswordLogin: !boolean("TERDUT_PASSWORD_LOGIN", true),
|
DisablePasswordLogin: !boolean("TERDUT_PASSWORD_LOGIN", true),
|
||||||
OIDC: loadOIDC(),
|
|
||||||
|
OperatorKey: strings.TrimSpace(os.Getenv("TERDUT_OPERATOR_KEY")),
|
||||||
|
TrustedProxies: integer("TERDUT_TRUSTED_PROXIES", 1),
|
||||||
|
OIDC: loadOIDC(),
|
||||||
|
|
||||||
OperatorMode: boolean("TERDUT_OPERATOR_MODE", false),
|
OperatorMode: boolean("TERDUT_OPERATOR_MODE", false),
|
||||||
}
|
}
|
||||||
@@ -187,6 +177,9 @@ func loadOIDC() OIDC {
|
|||||||
// provider would come up and then fail every login, which is harder to notice
|
// provider would come up and then fail every login, which is harder to notice
|
||||||
// than not starting.
|
// than not starting.
|
||||||
func (c Config) Validate() error {
|
func (c Config) Validate() error {
|
||||||
|
if c.OperatorKey != "" && len(c.OperatorKey) < MinOperatorKeyLength {
|
||||||
|
return fmt.Errorf("TERDUT_OPERATOR_KEY must be at least %d characters", MinOperatorKeyLength)
|
||||||
|
}
|
||||||
o := c.OIDC
|
o := c.OIDC
|
||||||
if !o.Enabled() {
|
if !o.Enabled() {
|
||||||
if c.DisablePasswordLogin {
|
if c.DisablePasswordLogin {
|
||||||
@@ -226,6 +219,16 @@ func str(env, def string) string {
|
|||||||
return def
|
return def
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// integer reads a non-negative int env var; anything else takes the default.
|
||||||
|
func integer(env string, def int) int {
|
||||||
|
if s := os.Getenv(env); s != "" {
|
||||||
|
if n, err := strconv.Atoi(strings.TrimSpace(s)); err == nil && n >= 0 {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return def
|
||||||
|
}
|
||||||
|
|
||||||
// list reads a comma- or space-separated env var.
|
// list reads a comma- or space-separated env var.
|
||||||
func list(env, def string) []string {
|
func list(env, def string) []string {
|
||||||
s := os.Getenv(env)
|
s := os.Getenv(env)
|
||||||
|
|||||||
@@ -35,8 +35,7 @@ const (
|
|||||||
//
|
//
|
||||||
// The pool is modest on purpose: this server's concurrency comes from a handful
|
// The pool is modest on purpose: this server's concurrency comes from a handful
|
||||||
// of HTTP handlers plus two background loops, and a cloud-native-pg instance
|
// of HTTP handlers plus two background loops, and a cloud-native-pg instance
|
||||||
// sized for it has a low max_connections. It is still a pool, unlike the single
|
// sized for it has a low max_connections.
|
||||||
// connection SQLite forced, so the notifier no longer blocks a webhook.
|
|
||||||
func Open(dsn string) (*sql.DB, error) {
|
func Open(dsn string) (*sql.DB, error) {
|
||||||
if dsn == "" {
|
if dsn == "" {
|
||||||
return nil, fmt.Errorf("empty DSN: set TERDUT_DB_DSN")
|
return nil, fmt.Errorf("empty DSN: set TERDUT_DB_DSN")
|
||||||
@@ -76,9 +75,8 @@ const migrationLockKey int64 = 7265_0003
|
|||||||
// Migrate applies every embedded migration that has not been applied yet, in
|
// Migrate applies every embedded migration that has not been applied yet, in
|
||||||
// filename order, recording each in schema_migrations.
|
// filename order, recording each in schema_migrations.
|
||||||
//
|
//
|
||||||
// Each file runs inside a transaction, which SQLite's version did not do: a
|
// Each file runs inside a transaction, so a migration that fails half way
|
||||||
// migration that failed half way used to leave the schema in whatever state it
|
// leaves the schema as it was: Postgres has transactional DDL.
|
||||||
// had reached. Postgres has transactional DDL, so the rollback is real.
|
|
||||||
func Migrate(db *sql.DB) error {
|
func Migrate(db *sql.DB) error {
|
||||||
ctx := context.Background()
|
ctx := context.Background()
|
||||||
conn, err := db.Conn(ctx)
|
conn, err := db.Conn(ctx)
|
||||||
|
|||||||
@@ -1,182 +0,0 @@
|
|||||||
-- The Postgres baseline: the schema as it stood at the end of the SQLite line,
|
|
||||||
-- in one file rather than ten.
|
|
||||||
--
|
|
||||||
-- The ten SQLite migrations are in git history up to the commit that introduced
|
|
||||||
-- this one, and they replay against nothing here: their shape was incremental
|
|
||||||
-- (columns added, then dropped again in 008) and 008's backfill rewrote data
|
|
||||||
-- that a Postgres install never had. An existing SQLite database is carried over
|
|
||||||
-- by scripts/sqlite-to-postgres.go, which copies rows into this schema.
|
|
||||||
--
|
|
||||||
-- Two conventions inherited deliberately:
|
|
||||||
--
|
|
||||||
-- * Timestamps are BIGINT unix seconds, not timestamptz. Everything in Go
|
|
||||||
-- already speaks epochs, and converting was a second change riding along
|
|
||||||
-- with the port. Worth revisiting on its own.
|
|
||||||
--
|
|
||||||
-- * Ids are GENERATED BY DEFAULT, not ALWAYS, so the migration script can
|
|
||||||
-- insert rows with their original ids and keep every foreign key intact.
|
|
||||||
-- setval at the end of the copy puts the sequences past them.
|
|
||||||
|
|
||||||
CREATE TABLE users (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
username TEXT NOT NULL UNIQUE,
|
|
||||||
email TEXT NOT NULL UNIQUE,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
-- Where this user's notifications go. NULL means they get none; incidents
|
|
||||||
-- assigned to them fall back to the configured fallback topic.
|
|
||||||
ntfy_topic TEXT,
|
|
||||||
-- NULL means the user has no password and can only use API keys.
|
|
||||||
password_hash TEXT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE api_keys (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
key_hash TEXT NOT NULL UNIQUE,
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
last_used_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
-- A session is a browser's credential, the cookie counterpart of an API key:
|
|
||||||
-- only the hash of the token is stored. expires_at slides forward while the
|
|
||||||
-- session is in use, so an on-call phone stays signed in.
|
|
||||||
CREATE TABLE sessions (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
token_hash TEXT NOT NULL UNIQUE,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
created_at BIGINT NOT NULL,
|
|
||||||
last_seen_at BIGINT NOT NULL,
|
|
||||||
expires_at BIGINT NOT NULL,
|
|
||||||
user_agent TEXT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_sessions_user ON sessions(user_id);
|
|
||||||
|
|
||||||
-- The machine-owned signal record: what Alertmanager says is true right now.
|
|
||||||
-- Workflow state lives on incidents, never here, because the webhook upsert owns
|
|
||||||
-- these rows and would overwrite it.
|
|
||||||
CREATE TABLE alerts (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
fingerprint TEXT NOT NULL UNIQUE,
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
status TEXT NOT NULL CHECK (status IN ('firing', 'resolved')),
|
|
||||||
labels JSONB NOT NULL DEFAULT '{}'::jsonb,
|
|
||||||
annotations JSONB NOT NULL DEFAULT '{}'::jsonb,
|
|
||||||
starts_at BIGINT NOT NULL,
|
|
||||||
ends_at BIGINT,
|
|
||||||
generator_url TEXT NOT NULL DEFAULT '',
|
|
||||||
received_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
archived_at BIGINT,
|
|
||||||
-- Why the alert left the firing state: 'alertmanager' when a resolved
|
|
||||||
-- webhook set it, 'expiry' when the sweeper inferred it from staleness.
|
|
||||||
resolution_source TEXT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX alerts_status_idx ON alerts(status);
|
|
||||||
CREATE INDEX alerts_name_idx ON alerts(name);
|
|
||||||
CREATE INDEX alerts_received_at_idx ON alerts(received_at DESC);
|
|
||||||
CREATE INDEX alerts_archived_at_idx ON alerts(archived_at);
|
|
||||||
|
|
||||||
CREATE TABLE schedule_entries (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
date TEXT NOT NULL UNIQUE, -- YYYY-MM-DD; one person per day
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX schedule_entries_date_idx ON schedule_entries(date);
|
|
||||||
|
|
||||||
-- The human work item: what people acknowledge, assign, snooze, discuss and
|
|
||||||
-- resolve. Correlation uses Alertmanager's own groupKey, so incidents follow the
|
|
||||||
-- group_by routing tree the operator already tuned.
|
|
||||||
CREATE TABLE incidents (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
group_key TEXT NOT NULL, -- Alertmanager groupKey, opaque
|
|
||||||
title TEXT NOT NULL, -- rendered from group_labels
|
|
||||||
group_labels JSONB NOT NULL DEFAULT '{}'::jsonb,
|
|
||||||
status TEXT NOT NULL CHECK (status IN ('triggered', 'acknowledged', 'resolved')),
|
|
||||||
severity TEXT, -- highest `severity` label across firing members
|
|
||||||
triggered_at BIGINT NOT NULL,
|
|
||||||
acknowledged_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
acknowledged_at BIGINT,
|
|
||||||
assigned_to BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
snoozed_until BIGINT,
|
|
||||||
resolved_at BIGINT,
|
|
||||||
resolution_source TEXT, -- 'alerts' | 'manual'
|
|
||||||
archived_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
-- Load-bearing: at most one OPEN incident per group_key. This is what makes
|
|
||||||
-- "resolved incident + a new alert occurrence = a new incident" work, and it is
|
|
||||||
-- the constraint the webhook's find-or-open lookup relies on.
|
|
||||||
CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents(group_key) WHERE resolved_at IS NULL;
|
|
||||||
CREATE INDEX incidents_status_idx ON incidents(status);
|
|
||||||
CREATE INDEX incidents_triggered_at_idx ON incidents(triggered_at DESC);
|
|
||||||
CREATE INDEX incidents_archived_at_idx ON incidents(archived_at);
|
|
||||||
|
|
||||||
-- Membership is historical, not a pointer on alerts: one alert row (one
|
|
||||||
-- fingerprint) resolves and re-fires over time and belongs to a different
|
|
||||||
-- incident each occurrence.
|
|
||||||
CREATE TABLE incident_alerts (
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
alert_id BIGINT NOT NULL REFERENCES alerts(id) ON DELETE CASCADE,
|
|
||||||
added_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
PRIMARY KEY (incident_id, alert_id)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts(alert_id);
|
|
||||||
|
|
||||||
-- The timeline. Append-only, and the only history this server keeps: alert rows
|
|
||||||
-- are mutated in place, so without this there is no record that anything
|
|
||||||
-- happened. Notes are events too, so one query renders the whole story.
|
|
||||||
CREATE TABLE incident_events (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
-- triggered | alert_added | alert_resolved | acknowledged | unacknowledged
|
|
||||||
-- | assigned | snoozed | unsnoozed | resolved | note | notified | notify_failed
|
|
||||||
type TEXT NOT NULL,
|
|
||||||
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, -- NULL = the server acted
|
|
||||||
alert_id BIGINT REFERENCES alerts(id) ON DELETE SET NULL,
|
|
||||||
detail TEXT,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incident_events_incident_idx ON incident_events(incident_id, created_at);
|
|
||||||
|
|
||||||
-- Delivery is an outbox rather than an inline HTTP call: a POST made while
|
|
||||||
-- holding the webhook's transaction would hold a connection open across a
|
|
||||||
-- network round trip. The webhook inserts a row; the notifier goroutine
|
|
||||||
-- delivers it.
|
|
||||||
CREATE TABLE notifications (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
-- Nullable: a notification sent to the fallback topic belongs to nobody,
|
|
||||||
-- because nobody was on call when the incident opened.
|
|
||||||
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
topic TEXT NOT NULL, -- resolved at enqueue: who was on call then
|
|
||||||
kind TEXT NOT NULL CHECK (kind IN ('triggered', 'reminder', 'resolved')),
|
|
||||||
created_at BIGINT NOT NULL,
|
|
||||||
send_after BIGINT NOT NULL, -- retry backoff watermark
|
|
||||||
attempts BIGINT NOT NULL DEFAULT 0,
|
|
||||||
sent_at BIGINT,
|
|
||||||
last_error TEXT -- kept after the last attempt, for debugging
|
|
||||||
);
|
|
||||||
|
|
||||||
-- The delivery loop's only query: what is due and still unsent.
|
|
||||||
CREATE INDEX notifications_pending_idx ON notifications(send_after) WHERE sent_at IS NULL;
|
|
||||||
-- Reminders and resolved notices both look up an incident's newest row.
|
|
||||||
CREATE INDEX notifications_incident_idx ON notifications(incident_id, id DESC);
|
|
||||||
|
|
||||||
-- A notification body is stored on the ntfy server and cached on the device, so
|
|
||||||
-- a real API key must never appear in one. Each delivery mints its own token
|
|
||||||
-- instead: one incident, one action, one day.
|
|
||||||
CREATE TABLE incident_ack_tokens (
|
|
||||||
token_hash TEXT PRIMARY KEY, -- SHA-256 of the raw token, as with api_keys
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
created_at BIGINT NOT NULL,
|
|
||||||
expires_at BIGINT NOT NULL
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens(expires_at);
|
|
||||||
@@ -0,0 +1,587 @@
|
|||||||
|
-- Terdut Server schema. One baseline: the project has not shipped, so the
|
||||||
|
-- migration history that led here (SQLite import, a Default team, per-team
|
||||||
|
-- deadman configs later replaced by switches) is not carried. Later changes are
|
||||||
|
-- new numbered files after this one.
|
||||||
|
--
|
||||||
|
-- Timestamps are Unix epoch seconds in BIGINT columns throughout.
|
||||||
|
|
||||||
|
CREATE TABLE alerts (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
fingerprint text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
status text NOT NULL,
|
||||||
|
labels jsonb DEFAULT '{}'::jsonb NOT NULL,
|
||||||
|
annotations jsonb DEFAULT '{}'::jsonb NOT NULL,
|
||||||
|
starts_at bigint NOT NULL,
|
||||||
|
ends_at bigint,
|
||||||
|
generator_url text DEFAULT ''::text NOT NULL,
|
||||||
|
received_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
archived_at bigint,
|
||||||
|
resolution_source text,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
integration_id bigint,
|
||||||
|
CONSTRAINT alerts_status_check CHECK ((status = ANY (ARRAY['firing'::text, 'resolved'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE api_keys (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
key_hash text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_used_at bigint,
|
||||||
|
expires_at bigint
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE deadman_switches (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
matcher text NOT NULL,
|
||||||
|
timeout_seconds bigint NOT NULL,
|
||||||
|
severity text DEFAULT 'critical'::text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
CONSTRAINT deadman_switches_timeout_seconds_check CHECK ((timeout_seconds > 0))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE device_logins (
|
||||||
|
device_hash text NOT NULL,
|
||||||
|
user_code text NOT NULL,
|
||||||
|
status text DEFAULT 'pending'::text NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
last_polled_at bigint DEFAULT 0 NOT NULL,
|
||||||
|
CONSTRAINT device_logins_status_check CHECK ((status = ANY (ARRAY['pending'::text, 'approved'::text, 'denied'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE escalation_levels (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
"position" bigint NOT NULL,
|
||||||
|
timeout_seconds bigint NOT NULL,
|
||||||
|
CONSTRAINT escalation_levels_timeout_seconds_check CHECK ((timeout_seconds > 0))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE escalation_policies (
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
repeat_count bigint DEFAULT 0 NOT NULL,
|
||||||
|
fallback_topic text DEFAULT ''::text NOT NULL,
|
||||||
|
updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
CONSTRAINT escalation_policies_repeat_count_check CHECK (((repeat_count >= 0) AND (repeat_count <= 10)))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE escalation_targets (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
level_id bigint NOT NULL,
|
||||||
|
kind text NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
CONSTRAINT escalation_targets_check CHECK ((((kind = 'user'::text) AND (user_id IS NOT NULL)) OR ((kind = 'oncall'::text) AND (user_id IS NULL)))),
|
||||||
|
CONSTRAINT escalation_targets_kind_check CHECK ((kind = ANY (ARRAY['user'::text, 'oncall'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incident_ack_tokens (
|
||||||
|
token_hash text NOT NULL,
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
created_at bigint NOT NULL,
|
||||||
|
expires_at bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incident_alerts (
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
alert_id bigint NOT NULL,
|
||||||
|
added_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incident_events (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
type text NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
alert_id bigint,
|
||||||
|
detail text,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
service_account_id bigint,
|
||||||
|
actor_user_id bigint,
|
||||||
|
actor_service_account_id bigint,
|
||||||
|
CONSTRAINT incident_events_actor_xor_chk CHECK (((user_id IS NULL) OR (service_account_id IS NULL))),
|
||||||
|
CONSTRAINT incident_events_assign_actor_xor_chk CHECK (((actor_user_id IS NULL) OR (actor_service_account_id IS NULL)))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incidents (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
group_key text NOT NULL,
|
||||||
|
title text NOT NULL,
|
||||||
|
group_labels jsonb DEFAULT '{}'::jsonb NOT NULL,
|
||||||
|
status text NOT NULL,
|
||||||
|
severity text,
|
||||||
|
triggered_at bigint NOT NULL,
|
||||||
|
acknowledged_by bigint,
|
||||||
|
acknowledged_at bigint,
|
||||||
|
assigned_to bigint,
|
||||||
|
snoozed_until bigint,
|
||||||
|
resolved_at bigint,
|
||||||
|
resolution_source text,
|
||||||
|
archived_at bigint,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
escalation_level bigint DEFAULT 0 NOT NULL,
|
||||||
|
escalation_level_at bigint,
|
||||||
|
escalation_round bigint DEFAULT 0 NOT NULL,
|
||||||
|
signature text DEFAULT ''::text NOT NULL,
|
||||||
|
acknowledged_by_service_account_id bigint,
|
||||||
|
CONSTRAINT incidents_ack_actor_xor_chk CHECK (((acknowledged_by IS NULL) OR (acknowledged_by_service_account_id IS NULL))),
|
||||||
|
CONSTRAINT incidents_status_check CHECK ((status = ANY (ARRAY['triggered'::text, 'acknowledged'::text, 'resolved'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE integrations (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
kind text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
key_hash text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_used_at bigint,
|
||||||
|
CONSTRAINT integrations_kind_check CHECK ((kind = 'alertmanager'::text))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE invites (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
token_hash text NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
role text NOT NULL,
|
||||||
|
created_by bigint,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
max_uses bigint DEFAULT 1 NOT NULL,
|
||||||
|
uses bigint DEFAULT 0 NOT NULL,
|
||||||
|
revoked_at bigint,
|
||||||
|
CONSTRAINT invites_max_uses_check CHECK (((max_uses > 0) AND (max_uses <= 100))),
|
||||||
|
CONSTRAINT invites_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE notifications (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
topic text NOT NULL,
|
||||||
|
kind text NOT NULL,
|
||||||
|
created_at bigint NOT NULL,
|
||||||
|
send_after bigint NOT NULL,
|
||||||
|
attempts bigint DEFAULT 0 NOT NULL,
|
||||||
|
sent_at bigint,
|
||||||
|
last_error text,
|
||||||
|
CONSTRAINT notifications_kind_check CHECK ((kind = ANY (ARRAY['triggered'::text, 'reminder'::text, 'resolved'::text, 'escalated'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE oidc_logins (
|
||||||
|
state_hash text NOT NULL,
|
||||||
|
nonce text NOT NULL,
|
||||||
|
pkce_verifier text NOT NULL,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
next text DEFAULT '/'::text NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE rate_limit_counters (
|
||||||
|
key text NOT NULL,
|
||||||
|
window_start bigint NOT NULL,
|
||||||
|
count integer NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE schedule_entries (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
date text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
team_id bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE service_account_keys (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
service_account_id bigint NOT NULL,
|
||||||
|
key_hash text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_used_at bigint
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE service_accounts (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
scope text NOT NULL,
|
||||||
|
team_id bigint,
|
||||||
|
created_by bigint,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
CONSTRAINT service_accounts_scope_check CHECK ((scope = ANY (ARRAY['instance'::text, 'team'::text]))),
|
||||||
|
CONSTRAINT service_accounts_scope_team_id_chk CHECK ((((scope = 'team'::text) AND (team_id IS NOT NULL)) OR ((scope = 'instance'::text) AND (team_id IS NULL))))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE sessions (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
token_hash text NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
created_at bigint NOT NULL,
|
||||||
|
last_seen_at bigint NOT NULL,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
user_agent text,
|
||||||
|
max_expires_at bigint
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE settings (
|
||||||
|
key text NOT NULL,
|
||||||
|
value text NOT NULL,
|
||||||
|
updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE team_members (
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
role text NOT NULL,
|
||||||
|
joined_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
source text DEFAULT 'manual'::text NOT NULL,
|
||||||
|
CONSTRAINT team_members_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text]))),
|
||||||
|
CONSTRAINT team_members_source_check CHECK ((source = ANY (ARRAY['manual'::text, 'oidc'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE teams (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
oidc_member_group text,
|
||||||
|
oidc_owner_group text,
|
||||||
|
-- A stable identity for a team managed by automation (terdut-operator:
|
||||||
|
-- "<namespace>/<name>" of its TerdutTeam), so it can find or recreate its own
|
||||||
|
-- team without trusting a display name. NULL for a team a person made.
|
||||||
|
external_id text
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE user_identities (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
issuer text NOT NULL,
|
||||||
|
subject text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_login_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE users (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
username text NOT NULL,
|
||||||
|
email text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
ntfy_topic text,
|
||||||
|
password_hash text,
|
||||||
|
is_admin boolean DEFAULT false NOT NULL,
|
||||||
|
disabled_at bigint,
|
||||||
|
invited_via bigint,
|
||||||
|
onboarding_dismissed_at bigint,
|
||||||
|
admin_source text DEFAULT 'manual'::text NOT NULL,
|
||||||
|
CONSTRAINT users_admin_source_check CHECK ((admin_source = ANY (ARRAY['manual'::text, 'oidc'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
ALTER TABLE alerts
|
||||||
|
ADD CONSTRAINT alerts_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE api_keys
|
||||||
|
ADD CONSTRAINT api_keys_key_hash_key UNIQUE (key_hash);
|
||||||
|
|
||||||
|
ALTER TABLE api_keys
|
||||||
|
ADD CONSTRAINT api_keys_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE deadman_switches
|
||||||
|
ADD CONSTRAINT deadman_switches_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE device_logins
|
||||||
|
ADD CONSTRAINT device_logins_pkey PRIMARY KEY (device_hash);
|
||||||
|
|
||||||
|
ALTER TABLE device_logins
|
||||||
|
ADD CONSTRAINT device_logins_user_code_key UNIQUE (user_code);
|
||||||
|
|
||||||
|
ALTER TABLE escalation_levels
|
||||||
|
ADD CONSTRAINT escalation_levels_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE escalation_levels
|
||||||
|
ADD CONSTRAINT escalation_levels_team_id_position_key UNIQUE (team_id, "position");
|
||||||
|
|
||||||
|
ALTER TABLE escalation_policies
|
||||||
|
ADD CONSTRAINT escalation_policies_pkey PRIMARY KEY (team_id);
|
||||||
|
|
||||||
|
ALTER TABLE escalation_targets
|
||||||
|
ADD CONSTRAINT escalation_targets_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE incident_ack_tokens
|
||||||
|
ADD CONSTRAINT incident_ack_tokens_pkey PRIMARY KEY (token_hash);
|
||||||
|
|
||||||
|
ALTER TABLE incident_alerts
|
||||||
|
ADD CONSTRAINT incident_alerts_pkey PRIMARY KEY (incident_id, alert_id);
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE integrations
|
||||||
|
ADD CONSTRAINT integrations_key_hash_key UNIQUE (key_hash);
|
||||||
|
|
||||||
|
ALTER TABLE integrations
|
||||||
|
ADD CONSTRAINT integrations_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_token_hash_key UNIQUE (token_hash);
|
||||||
|
|
||||||
|
ALTER TABLE notifications
|
||||||
|
ADD CONSTRAINT notifications_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE oidc_logins
|
||||||
|
ADD CONSTRAINT oidc_logins_pkey PRIMARY KEY (state_hash);
|
||||||
|
|
||||||
|
ALTER TABLE rate_limit_counters
|
||||||
|
ADD CONSTRAINT rate_limit_counters_pkey PRIMARY KEY (key);
|
||||||
|
|
||||||
|
ALTER TABLE schedule_entries
|
||||||
|
ADD CONSTRAINT schedule_entries_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE service_account_keys
|
||||||
|
ADD CONSTRAINT service_account_keys_key_hash_key UNIQUE (key_hash);
|
||||||
|
|
||||||
|
ALTER TABLE service_account_keys
|
||||||
|
ADD CONSTRAINT service_account_keys_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_name_key UNIQUE (name);
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE sessions
|
||||||
|
ADD CONSTRAINT sessions_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE sessions
|
||||||
|
ADD CONSTRAINT sessions_token_hash_key UNIQUE (token_hash);
|
||||||
|
|
||||||
|
ALTER TABLE settings
|
||||||
|
ADD CONSTRAINT settings_pkey PRIMARY KEY (key);
|
||||||
|
|
||||||
|
ALTER TABLE team_members
|
||||||
|
ADD CONSTRAINT team_members_pkey PRIMARY KEY (team_id, user_id);
|
||||||
|
|
||||||
|
ALTER TABLE teams
|
||||||
|
ADD CONSTRAINT teams_name_key UNIQUE (name);
|
||||||
|
|
||||||
|
ALTER TABLE teams
|
||||||
|
ADD CONSTRAINT teams_external_id_key UNIQUE (external_id);
|
||||||
|
|
||||||
|
ALTER TABLE teams
|
||||||
|
ADD CONSTRAINT teams_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE user_identities
|
||||||
|
ADD CONSTRAINT user_identities_issuer_subject_key UNIQUE (issuer, subject);
|
||||||
|
|
||||||
|
ALTER TABLE user_identities
|
||||||
|
ADD CONSTRAINT user_identities_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_email_key UNIQUE (email);
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_username_key UNIQUE (username);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_archived_at_idx ON alerts USING btree (archived_at);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_integration_idx ON alerts USING btree (integration_id, received_at) WHERE (integration_id IS NOT NULL);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_name_idx ON alerts USING btree (name);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_received_at_idx ON alerts USING btree (received_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_status_idx ON alerts USING btree (status);
|
||||||
|
|
||||||
|
CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts USING btree (team_id, fingerprint);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_team_received_idx ON alerts USING btree (team_id, received_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX deadman_switches_team_idx ON deadman_switches USING btree (team_id);
|
||||||
|
|
||||||
|
CREATE INDEX device_logins_expires_idx ON device_logins USING btree (expires_at);
|
||||||
|
|
||||||
|
CREATE INDEX escalation_targets_level_idx ON escalation_targets USING btree (level_id);
|
||||||
|
|
||||||
|
CREATE INDEX idx_sessions_user ON sessions USING btree (user_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens USING btree (expires_at);
|
||||||
|
|
||||||
|
CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts USING btree (alert_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events USING btree (actor_service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_actor_user_id_idx ON incident_events USING btree (actor_user_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_incident_idx ON incident_events USING btree (incident_id, created_at);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_service_account_id_idx ON incident_events USING btree (service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_acknowledged_by_service_account_id_idx ON incidents USING btree (acknowledged_by_service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_archived_at_idx ON incidents USING btree (archived_at);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_escalation_idx ON incidents USING btree (escalation_level_at) WHERE ((resolved_at IS NULL) AND (status = 'triggered'::text));
|
||||||
|
|
||||||
|
CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents USING btree (team_id, group_key) WHERE (resolved_at IS NULL);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_signature_idx ON incidents USING btree (team_id, signature, triggered_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_status_idx ON incidents USING btree (status);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_team_triggered_idx ON incidents USING btree (team_id, triggered_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_triggered_at_idx ON incidents USING btree (triggered_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX integrations_team_idx ON integrations USING btree (team_id);
|
||||||
|
|
||||||
|
-- A name identifies an integration (and a switch) within its team, so a client
|
||||||
|
-- that manages them declaratively can look one up by name instead of listing
|
||||||
|
-- and matching.
|
||||||
|
CREATE UNIQUE INDEX integrations_team_name_key ON integrations (team_id, name);
|
||||||
|
CREATE UNIQUE INDEX deadman_switches_team_name_key ON deadman_switches (team_id, name);
|
||||||
|
|
||||||
|
CREATE INDEX invites_team_idx ON invites USING btree (team_id);
|
||||||
|
|
||||||
|
CREATE INDEX notifications_incident_idx ON notifications USING btree (incident_id, id DESC);
|
||||||
|
|
||||||
|
CREATE INDEX notifications_pending_idx ON notifications USING btree (send_after) WHERE (sent_at IS NULL);
|
||||||
|
|
||||||
|
CREATE INDEX oidc_logins_expires_idx ON oidc_logins USING btree (expires_at);
|
||||||
|
|
||||||
|
CREATE INDEX schedule_entries_date_idx ON schedule_entries USING btree (date);
|
||||||
|
|
||||||
|
CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries USING btree (team_id, date);
|
||||||
|
|
||||||
|
CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys USING btree (service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX service_accounts_team_id_idx ON service_accounts USING btree (team_id);
|
||||||
|
|
||||||
|
CREATE INDEX team_members_user_idx ON team_members USING btree (user_id);
|
||||||
|
|
||||||
|
CREATE INDEX user_identities_user_idx ON user_identities USING btree (user_id);
|
||||||
|
|
||||||
|
CREATE INDEX users_is_admin_idx ON users USING btree (is_admin) WHERE is_admin;
|
||||||
|
|
||||||
|
ALTER TABLE alerts
|
||||||
|
ADD CONSTRAINT alerts_integration_id_fkey FOREIGN KEY (integration_id) REFERENCES integrations(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE alerts
|
||||||
|
ADD CONSTRAINT alerts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE api_keys
|
||||||
|
ADD CONSTRAINT api_keys_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE deadman_switches
|
||||||
|
ADD CONSTRAINT deadman_switches_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE device_logins
|
||||||
|
ADD CONSTRAINT device_logins_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_levels
|
||||||
|
ADD CONSTRAINT escalation_levels_team_id_fkey FOREIGN KEY (team_id) REFERENCES escalation_policies(team_id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_policies
|
||||||
|
ADD CONSTRAINT escalation_policies_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_targets
|
||||||
|
ADD CONSTRAINT escalation_targets_level_id_fkey FOREIGN KEY (level_id) REFERENCES escalation_levels(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_targets
|
||||||
|
ADD CONSTRAINT escalation_targets_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_ack_tokens
|
||||||
|
ADD CONSTRAINT incident_ack_tokens_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_ack_tokens
|
||||||
|
ADD CONSTRAINT incident_ack_tokens_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_alerts
|
||||||
|
ADD CONSTRAINT incident_alerts_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_alerts
|
||||||
|
ADD CONSTRAINT incident_alerts_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_actor_service_account_id_fkey FOREIGN KEY (actor_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_actor_user_id_fkey FOREIGN KEY (actor_user_id) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_acknowledged_by_fkey FOREIGN KEY (acknowledged_by) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_acknowledged_by_service_account_id_fkey FOREIGN KEY (acknowledged_by_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_assigned_to_fkey FOREIGN KEY (assigned_to) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE integrations
|
||||||
|
ADD CONSTRAINT integrations_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE notifications
|
||||||
|
ADD CONSTRAINT notifications_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE notifications
|
||||||
|
ADD CONSTRAINT notifications_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE schedule_entries
|
||||||
|
ADD CONSTRAINT schedule_entries_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE schedule_entries
|
||||||
|
ADD CONSTRAINT schedule_entries_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE service_account_keys
|
||||||
|
ADD CONSTRAINT service_account_keys_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE sessions
|
||||||
|
ADD CONSTRAINT sessions_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE team_members
|
||||||
|
ADD CONSTRAINT team_members_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE team_members
|
||||||
|
ADD CONSTRAINT team_members_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE user_identities
|
||||||
|
ADD CONSTRAINT user_identities_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_invited_via_fkey FOREIGN KEY (invited_via) REFERENCES invites(id) ON DELETE SET NULL;
|
||||||
@@ -1,25 +0,0 @@
|
|||||||
-- A system administrator role, and the first thing in this server that one user
|
|
||||||
-- can do and another cannot.
|
|
||||||
--
|
|
||||||
-- Until now every authenticated caller could create and delete users, set
|
|
||||||
-- anybody's password and mint API keys for anybody — auth.go said so in a
|
|
||||||
-- comment. That was defensible with one operator and a hand-made account; it is
|
|
||||||
-- not once people sign themselves up (see #7).
|
|
||||||
--
|
|
||||||
-- EVERY EXISTING USER BECOMES AN ADMIN. They already hold these powers, so
|
|
||||||
-- this migration changes nobody's access: it names what is already true, and
|
|
||||||
-- leaves demotion as a deliberate act somebody performs afterwards. The
|
|
||||||
-- alternative — promoting only user 1 — would silently strip the others, and
|
|
||||||
-- could leave an install whose only admin is an account nobody has a password
|
|
||||||
-- for.
|
|
||||||
--
|
|
||||||
-- New users are not admins: the column defaults to false, and the only ways to
|
|
||||||
-- become one are this backfill, the bootstrap endpoint, or an existing admin
|
|
||||||
-- granting it.
|
|
||||||
ALTER TABLE users ADD COLUMN is_admin BOOLEAN NOT NULL DEFAULT false;
|
|
||||||
|
|
||||||
UPDATE users SET is_admin = true;
|
|
||||||
|
|
||||||
-- The queue's assignment dropdown and the on-call schedule read every user, and
|
|
||||||
-- the admin screens in #5 will filter on this.
|
|
||||||
CREATE INDEX users_is_admin_idx ON users(is_admin) WHERE is_admin;
|
|
||||||
@@ -1,103 +0,0 @@
|
|||||||
-- Teams: the unit of tenancy. Everything a person works on now belongs to one.
|
|
||||||
--
|
|
||||||
-- Until this migration the install was one shared space — every user saw every
|
|
||||||
-- alert and every incident, and the Alertmanager webhook was unauthenticated, so
|
|
||||||
-- anything that could reach the port could open an incident for everybody.
|
|
||||||
--
|
|
||||||
-- The shape, in one paragraph: a team owns its incidents, alerts, schedule and
|
|
||||||
-- integrations. A user belongs to as many teams as they like, with a role in
|
|
||||||
-- each: an `owner` configures the team, a `member` works its incidents. An
|
|
||||||
-- integration key is what an alert arrives on, and the key is what says which
|
|
||||||
-- team the alert belongs to.
|
|
||||||
--
|
|
||||||
-- EVERYTHING EXISTING MOVES INTO ONE DEFAULT TEAM, and every existing user
|
|
||||||
-- becomes an owner of it. That keeps an upgrade a no-op for the people using it:
|
|
||||||
-- the same queue, the same schedule, the same incidents, with a name on them.
|
|
||||||
|
|
||||||
CREATE TABLE teams (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
name TEXT NOT NULL UNIQUE,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
-- role is free text with a CHECK rather than an enum, so adding a third role
|
|
||||||
-- later is a migration and not a type rewrite.
|
|
||||||
CREATE TABLE team_members (
|
|
||||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
|
|
||||||
joined_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
PRIMARY KEY (team_id, user_id)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX team_members_user_idx ON team_members(user_id);
|
|
||||||
|
|
||||||
-- How alerts get in, and the only thing that says which team they belong to.
|
|
||||||
-- The key is stored as a SHA-256 hash, like api_keys and the ack tokens: a
|
|
||||||
-- leaked database gives nobody the ability to post alerts.
|
|
||||||
CREATE TABLE integrations (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
kind TEXT NOT NULL CHECK (kind IN ('alertmanager')),
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
key_hash TEXT NOT NULL UNIQUE,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
last_used_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX integrations_team_idx ON integrations(team_id);
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- The default team, and everything that already exists moving into it.
|
|
||||||
--
|
|
||||||
-- Created unconditionally, even on an empty install, so there is always a team
|
|
||||||
-- for the bootstrap user to land in and for the first integration to hang off.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
INSERT INTO teams (name) VALUES ('Default');
|
|
||||||
|
|
||||||
INSERT INTO team_members (team_id, user_id, role)
|
|
||||||
SELECT (SELECT id FROM teams WHERE name = 'Default'), id, 'owner' FROM users;
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- team_id on everything a team owns.
|
|
||||||
--
|
|
||||||
-- Added nullable, backfilled, then made NOT NULL: adding a NOT NULL column with
|
|
||||||
-- no default to a table with rows is rejected, and a DEFAULT pointing at the
|
|
||||||
-- default team would quietly keep working after the default team is gone.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
ALTER TABLE alerts ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
|
||||||
ALTER TABLE incidents ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
|
||||||
ALTER TABLE schedule_entries ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
|
||||||
|
|
||||||
UPDATE alerts SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
|
||||||
UPDATE incidents SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
|
||||||
UPDATE schedule_entries SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
|
||||||
|
|
||||||
ALTER TABLE alerts ALTER COLUMN team_id SET NOT NULL;
|
|
||||||
ALTER TABLE incidents ALTER COLUMN team_id SET NOT NULL;
|
|
||||||
ALTER TABLE schedule_entries ALTER COLUMN team_id SET NOT NULL;
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- The uniqueness rules were all written for one tenant, and every one of them
|
|
||||||
-- is wrong now: two teams monitoring two clusters legitimately see the same
|
|
||||||
-- fingerprint, the same groupKey, and want somebody on call on the same day.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
ALTER TABLE alerts DROP CONSTRAINT alerts_fingerprint_key;
|
|
||||||
CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts(team_id, fingerprint);
|
|
||||||
|
|
||||||
DROP INDEX incidents_open_group_key_idx;
|
|
||||||
-- Still load-bearing, now per team: at most one OPEN incident per group_key
|
|
||||||
-- within a team. This is what makes "resolved incident + a new alert occurrence
|
|
||||||
-- = a new incident" work, and what the webhook's find-or-open lookup relies on.
|
|
||||||
CREATE UNIQUE INDEX incidents_open_group_key_idx
|
|
||||||
ON incidents(team_id, group_key) WHERE resolved_at IS NULL;
|
|
||||||
|
|
||||||
ALTER TABLE schedule_entries DROP CONSTRAINT schedule_entries_date_key;
|
|
||||||
CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries(team_id, date);
|
|
||||||
|
|
||||||
-- The list views all filter by team first.
|
|
||||||
CREATE INDEX alerts_team_received_idx ON alerts(team_id, received_at DESC);
|
|
||||||
CREATE INDEX incidents_team_triggered_idx ON incidents(team_id, triggered_at DESC);
|
|
||||||
@@ -1,39 +0,0 @@
|
|||||||
-- Dead man's switches become a team's own configuration.
|
|
||||||
--
|
|
||||||
-- They were three environment variables — TERDUT_DEADMAN_MATCHERS, _TIMEOUT and
|
|
||||||
-- _SEVERITY — which made them one setting for the whole install. That was the
|
|
||||||
-- last piece of the alerting path a team could not control: a team could take
|
|
||||||
-- its own alerts on its own key and still not say which of them were
|
|
||||||
-- heartbeats, or how long a silence had to last before somebody was paged.
|
|
||||||
--
|
|
||||||
-- One row per team rather than one row per switch. The unit of monitoring is
|
|
||||||
-- still the fingerprint, as it always was — two clusters sending the same
|
|
||||||
-- heartbeat alertname are two independent switches — and the matcher string
|
|
||||||
-- keeps the format the environment variable used, so a value can be moved from
|
|
||||||
-- one to the other unchanged.
|
|
||||||
--
|
|
||||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
|
||||||
-- inserts a row per team at startup from its own configuration, and the same
|
|
||||||
-- values therefore carry forward into the first team's row without anybody
|
|
||||||
-- retyping them. See seedDeadmanConfigs.
|
|
||||||
CREATE TABLE deadman_configs (
|
|
||||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
|
|
||||||
-- ";" separates matchers, "," the label conditions within one, "=" is exact
|
|
||||||
-- equality: `alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat`.
|
|
||||||
-- Every matcher must name an alertname. Empty watches nothing.
|
|
||||||
matchers TEXT NOT NULL DEFAULT '',
|
|
||||||
|
|
||||||
-- Seconds rather than a Go duration string: the column is compared and
|
|
||||||
-- arithmetic is done on it, and a value that has to be parsed before it can
|
|
||||||
-- be believed is a value that can be stored unparseable. Zero disables the
|
|
||||||
-- team's switches entirely.
|
|
||||||
timeout_seconds BIGINT NOT NULL DEFAULT 0,
|
|
||||||
|
|
||||||
-- The severity these incidents open at. They have no member alerts to
|
|
||||||
-- derive one from, and a heartbeat's own severity label is meaningless —
|
|
||||||
-- Watchdog ships as "none".
|
|
||||||
severity TEXT NOT NULL DEFAULT 'critical',
|
|
||||||
|
|
||||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
@@ -1,35 +0,0 @@
|
|||||||
-- Settings that an administrator can change without a redeploy, and the flag
|
|
||||||
-- that takes an account out of use without deleting it.
|
|
||||||
--
|
|
||||||
-- Three of the server's tunables were environment variables, which meant
|
|
||||||
-- changing how long an incident waits before it is paged again required editing
|
|
||||||
-- a chart, merging it, and waiting for a reconcile. They are behaviour, not
|
|
||||||
-- infrastructure, and the difference is who needs to change them and how often.
|
|
||||||
--
|
|
||||||
-- What stays in the environment: the ntfy URL and token, the database DSN, the
|
|
||||||
-- listen address and the public URL. Those are where the server is plugged in
|
|
||||||
-- rather than how it behaves, they are needed before the database is open, and
|
|
||||||
-- two of them are credentials.
|
|
||||||
--
|
|
||||||
-- Key/value rather than a column per setting. A settings table with one row and
|
|
||||||
-- a column per knob needs a migration for every new knob, and #6 and #7 will
|
|
||||||
-- both add some. The cost is that values are text and the accessor has to say
|
|
||||||
-- what type it wanted; settings.go does that in one place.
|
|
||||||
--
|
|
||||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
|
||||||
-- inserts each key from its own configuration at startup, once, so an install
|
|
||||||
-- that upgrades keeps exactly the behaviour it had. See SeedSettings.
|
|
||||||
CREATE TABLE settings (
|
|
||||||
key TEXT PRIMARY KEY,
|
|
||||||
value TEXT NOT NULL,
|
|
||||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
-- Disabling an account rather than deleting it: the person has left, or the
|
|
||||||
-- credential is suspect, and their incidents, acknowledgements and timeline
|
|
||||||
-- entries must stay exactly where they are. Deleting a user nulls their
|
|
||||||
-- acknowledged_by and assigned_to, which quietly rewrites history.
|
|
||||||
--
|
|
||||||
-- A disabled user cannot sign in and their API keys stop working, but they are
|
|
||||||
-- still a name the timeline can show and still a member of their teams.
|
|
||||||
ALTER TABLE users ADD COLUMN disabled_at BIGINT;
|
|
||||||
@@ -1,95 +0,0 @@
|
|||||||
-- Escalation: page somebody else when the first person does not answer.
|
|
||||||
--
|
|
||||||
-- This is the gap the whole multi-tenancy line of work was opened to close.
|
|
||||||
-- Until now an unacknowledged incident re-paged the same topic every
|
|
||||||
-- notify_repeat forever, which is a louder version of the same silence: if the
|
|
||||||
-- person on call is asleep, has no signal, or has left, nothing else happens.
|
|
||||||
--
|
|
||||||
-- Shape: one policy per team, an ordered list of levels, each level with a
|
|
||||||
-- timeout and a set of targets. When a level's timeout passes and the incident
|
|
||||||
-- is still triggered, the next level is paged. When the last level passes, the
|
|
||||||
-- chain repeats repeat_count times, and then the team's fallback topic is paged
|
|
||||||
-- once as the end of the line.
|
|
||||||
--
|
|
||||||
-- A team WITHOUT a policy keeps exactly today's behaviour: page the assignee,
|
|
||||||
-- then remind on the same topic. Escalation is opt-in per team, and the two
|
|
||||||
-- never both run for one incident -- see enqueueReminders.
|
|
||||||
CREATE TABLE escalation_policies (
|
|
||||||
-- One per team for now, hence the team as the key rather than an id with a
|
|
||||||
-- unique index: routing different alerts to different chains needs the
|
|
||||||
-- alert to carry something to route ON, which is a separate question.
|
|
||||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
|
|
||||||
-- How many extra times to run the whole chain after it has been walked
|
|
||||||
-- once. 0 means walk it once and stop at the fallback.
|
|
||||||
repeat_count BIGINT NOT NULL DEFAULT 0 CHECK (repeat_count >= 0 AND repeat_count <= 10),
|
|
||||||
|
|
||||||
-- Where the last page goes when every level has been tried. Per team now:
|
|
||||||
-- TERDUT_NTFY_FALLBACK_TOPIC was one topic for the whole install, which in
|
|
||||||
-- a multi-team server pages the wrong people. Empty means the chain simply
|
|
||||||
-- ends.
|
|
||||||
fallback_topic TEXT NOT NULL DEFAULT '',
|
|
||||||
|
|
||||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE escalation_levels (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
team_id BIGINT NOT NULL REFERENCES escalation_policies(team_id) ON DELETE CASCADE,
|
|
||||||
-- 1-based, dense. The API rewrites the whole ladder on every edit rather
|
|
||||||
-- than patching one rung, so there is no way to leave a gap.
|
|
||||||
position BIGINT NOT NULL,
|
|
||||||
-- How long this level has to produce an acknowledgement before the next one
|
|
||||||
-- is paged. Seconds, like every other duration in this schema.
|
|
||||||
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
|
|
||||||
|
|
||||||
UNIQUE (team_id, position)
|
|
||||||
);
|
|
||||||
|
|
||||||
-- Who a level pages. Either a named person, or whoever the team's rota says is
|
|
||||||
-- on call today -- which is the target that keeps working when the rota
|
|
||||||
-- changes and nobody remembers to edit the policy.
|
|
||||||
CREATE TABLE escalation_targets (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
level_id BIGINT NOT NULL REFERENCES escalation_levels(id) ON DELETE CASCADE,
|
|
||||||
kind TEXT NOT NULL CHECK (kind IN ('user', 'oncall')),
|
|
||||||
-- Set for kind='user', NULL for kind='oncall'.
|
|
||||||
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
|
|
||||||
CHECK ((kind = 'user' AND user_id IS NOT NULL) OR (kind = 'oncall' AND user_id IS NULL))
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX escalation_targets_level_idx ON escalation_targets(level_id);
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- Where an incident is in its chain.
|
|
||||||
--
|
|
||||||
-- On the incident rather than in a side table: it is read on every notifier
|
|
||||||
-- tick alongside the incident's status, and one row per incident is exactly
|
|
||||||
-- what the state is.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
-- 0 means no level has been paged yet, which is the state of every incident
|
|
||||||
-- that existed before escalation and of every incident in a team with no
|
|
||||||
-- policy. 1 is the first level.
|
|
||||||
ALTER TABLE incidents ADD COLUMN escalation_level BIGINT NOT NULL DEFAULT 0;
|
|
||||||
|
|
||||||
-- When the current level was entered, and therefore what its timeout is
|
|
||||||
-- measured from. NULL while escalation_level is 0.
|
|
||||||
ALTER TABLE incidents ADD COLUMN escalation_level_at BIGINT;
|
|
||||||
|
|
||||||
-- How many times the chain has been walked in full. Compared against the
|
|
||||||
-- policy's repeat_count.
|
|
||||||
ALTER TABLE incidents ADD COLUMN escalation_round BIGINT NOT NULL DEFAULT 0;
|
|
||||||
|
|
||||||
-- The notifier's escalation query: incidents still waiting, oldest level first.
|
|
||||||
CREATE INDEX incidents_escalation_idx
|
|
||||||
ON incidents(escalation_level_at)
|
|
||||||
WHERE resolved_at IS NULL AND status = 'triggered';
|
|
||||||
|
|
||||||
-- 'escalated' joins the outbox kinds: a page that went out because nobody
|
|
||||||
-- answered the last one, which is worth telling apart from the first page and
|
|
||||||
-- from a reminder when reading the timeline or debugging a delivery.
|
|
||||||
ALTER TABLE notifications DROP CONSTRAINT notifications_kind_check;
|
|
||||||
ALTER TABLE notifications ADD CONSTRAINT notifications_kind_check
|
|
||||||
CHECK (kind IN ('triggered', 'reminder', 'resolved', 'escalated'));
|
|
||||||
@@ -1,49 +0,0 @@
|
|||||||
-- Self-service sign-up, and the invite links that make it useful.
|
|
||||||
--
|
|
||||||
-- Until now the only way to get an account was for somebody who already had one
|
|
||||||
-- to create it, and the login page told people to "ask an admin". That is a
|
|
||||||
-- workable arrangement for one operator and an impossible one for a team.
|
|
||||||
--
|
|
||||||
-- An invite is a link, not an email: this server has no SMTP and adding it to
|
|
||||||
-- send one message would be a new subsystem to run, secure and monitor. The
|
|
||||||
-- person inviting sends the link however they already talk to the person they
|
|
||||||
-- are inviting.
|
|
||||||
CREATE TABLE invites (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
|
|
||||||
-- SHA-256 of the raw token, like api_keys, the integration keys and the
|
|
||||||
-- acknowledgement tokens. A leaked database hands nobody an account.
|
|
||||||
token_hash TEXT NOT NULL UNIQUE,
|
|
||||||
|
|
||||||
-- Which team the invitee lands in, and as what. An invite always names a
|
|
||||||
-- team: an account in no team sees an empty queue and can be paged by
|
|
||||||
-- nobody, which is not a state to invite somebody into.
|
|
||||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
|
|
||||||
|
|
||||||
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
|
|
||||||
-- Invites expire. A link that works forever is a credential nobody
|
|
||||||
-- remembers issuing, sitting in a chat log.
|
|
||||||
expires_at BIGINT NOT NULL,
|
|
||||||
|
|
||||||
-- Single-use by default: max_uses 1. A team onboarding six people at once
|
|
||||||
-- can raise it rather than minting six links.
|
|
||||||
max_uses BIGINT NOT NULL DEFAULT 1 CHECK (max_uses > 0 AND max_uses <= 100),
|
|
||||||
uses BIGINT NOT NULL DEFAULT 0,
|
|
||||||
|
|
||||||
-- Revoked by hand, separately from expiry, so "this link is no longer
|
|
||||||
-- wanted" and "this link timed out" stay distinguishable in the listing.
|
|
||||||
revoked_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX invites_team_idx ON invites(team_id);
|
|
||||||
|
|
||||||
-- Who redeemed which invite. Kept after the invite is gone — the answer to "how
|
|
||||||
-- did this account get here" should outlive the link that made it.
|
|
||||||
ALTER TABLE users ADD COLUMN invited_via BIGINT REFERENCES invites(id) ON DELETE SET NULL;
|
|
||||||
|
|
||||||
-- Where a person is in the first-run checklist, so it can be resumed and
|
|
||||||
-- dismissed rather than nagging forever. One row per user, created on demand.
|
|
||||||
ALTER TABLE users ADD COLUMN onboarding_dismissed_at BIGINT;
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
-- Similar incidents: a signature per incident, so "has this happened before"
|
|
||||||
-- is an indexed equality instead of a search.
|
|
||||||
--
|
|
||||||
-- The signature is the alert name plus the group labels that identify WHAT is
|
|
||||||
-- broken, minus the ones that only say WHERE it happened to run this time
|
|
||||||
-- (instance, pod, ...). Two incidents with the same signature in the same team
|
|
||||||
-- are the same problem for a responder's purposes.
|
|
||||||
--
|
|
||||||
-- Computed in Go for new incidents (incidentSignature in incident_store.go).
|
|
||||||
-- The backfill below MUST produce the same string; keep the volatile list in
|
|
||||||
-- both places in step.
|
|
||||||
ALTER TABLE incidents ADD COLUMN signature TEXT NOT NULL DEFAULT '';
|
|
||||||
|
|
||||||
UPDATE incidents SET signature =
|
|
||||||
COALESCE(NULLIF(group_labels->>'alertname', ''), title) || '|' ||
|
|
||||||
COALESCE((
|
|
||||||
SELECT string_agg(e.k || '=' || e.v, ',' ORDER BY e.k)
|
|
||||||
FROM jsonb_each_text(incidents.group_labels) AS e(k, v)
|
|
||||||
WHERE e.k <> 'alertname'
|
|
||||||
AND e.k NOT IN ('instance', 'pod', 'pod_name', 'pod_ip', 'container', 'container_name', 'endpoint')
|
|
||||||
), '');
|
|
||||||
|
|
||||||
CREATE INDEX incidents_signature_idx ON incidents(team_id, signature, triggered_at DESC);
|
|
||||||
@@ -1,54 +0,0 @@
|
|||||||
-- Dead man's switches become rows of their own.
|
|
||||||
--
|
|
||||||
-- 004 kept a team's switches in one string with one timeout and one severity,
|
|
||||||
-- which was enough to configure them and not enough to show them: there was no
|
|
||||||
-- thing to list, nothing to hang a status on, and every switch in a team had to
|
|
||||||
-- share a deadline. A row per switch gives each its own name, matcher, timeout
|
|
||||||
-- and severity, and gives the Team → Switches page something to be a list of.
|
|
||||||
--
|
|
||||||
-- The matcher keeps the syntax the string used, one matcher per row:
|
|
||||||
-- `alertname=Watchdog,cluster=prod`. The unit of monitoring is still the
|
|
||||||
-- fingerprint, so a matcher that many clusters satisfy is still one switch row
|
|
||||||
-- watching several independent heartbeats.
|
|
||||||
CREATE TABLE deadman_switches (
|
|
||||||
id BIGSERIAL PRIMARY KEY,
|
|
||||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
|
|
||||||
-- What the owner calls it. Defaults to the matcher when they do not say.
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
|
|
||||||
-- "," separates the label conditions, "=" is exact equality, and alertname is
|
|
||||||
-- mandatory: it is what keeps the sweeper's candidate query on an index.
|
|
||||||
matcher TEXT NOT NULL,
|
|
||||||
|
|
||||||
-- Seconds of silence before the switch is declared dead. Never zero: a switch
|
|
||||||
-- that cannot fire is deleted, not disabled.
|
|
||||||
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
|
|
||||||
|
|
||||||
-- The severity its incidents open at. See 004 for why they carry their own.
|
|
||||||
severity TEXT NOT NULL DEFAULT 'critical',
|
|
||||||
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX deadman_switches_team_idx ON deadman_switches (team_id);
|
|
||||||
|
|
||||||
-- Carry every team's configuration over, one row per matcher. A team whose
|
|
||||||
-- timeout was zero had switches turned off, which is now "no rows".
|
|
||||||
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
|
||||||
SELECT c.team_id, btrim(m), btrim(m), c.timeout_seconds, c.severity
|
|
||||||
FROM deadman_configs c,
|
|
||||||
LATERAL regexp_split_to_table(c.matchers, ';') AS m
|
|
||||||
WHERE c.timeout_seconds > 0
|
|
||||||
AND btrim(m) <> ''
|
|
||||||
ORDER BY c.team_id;
|
|
||||||
|
|
||||||
-- The server seeds environment defaults into teams once, and remembers that it
|
|
||||||
-- did. An install that had a row per team was already seeded; without this
|
|
||||||
-- marker the first start after upgrading would seed teams that had switched
|
|
||||||
-- theirs off.
|
|
||||||
INSERT INTO settings (key, value)
|
|
||||||
SELECT 'deadman_seeded', '1'
|
|
||||||
WHERE EXISTS (SELECT 1 FROM deadman_configs);
|
|
||||||
|
|
||||||
DROP TABLE deadman_configs;
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
-- Which alert source an alert last arrived on.
|
|
||||||
--
|
|
||||||
-- Team -> Sources shows when each source last posted, which integrations
|
|
||||||
-- already knew (last_used_at, stamped on every webhook). What it could not say
|
|
||||||
-- was what a source delivered: an alert never recorded the key it came in on, so
|
|
||||||
-- "prod alertmanager" and "staging alertmanager" were indistinguishable once
|
|
||||||
-- inside. This column is that link, and lets the page show each source's last
|
|
||||||
-- alert and how many alerts it has kept fresh over the past day.
|
|
||||||
--
|
|
||||||
-- Last sender wins: every accepted payload restamps it, the way it advances
|
|
||||||
-- received_at. Two sources posting the same fingerprint into one team is
|
|
||||||
-- already one alert, and it is attributed to whichever spoke last.
|
|
||||||
--
|
|
||||||
-- Nullable, and not backfilled. Alerts that arrived before this migration have
|
|
||||||
-- no source, and NULL says so honestly rather than guessing. It heals by itself:
|
|
||||||
-- Alertmanager re-sends every alert each repeat_interval, and each re-send is an
|
|
||||||
-- accepted payload. Deleting a source keeps its alerts, unattributed.
|
|
||||||
ALTER TABLE alerts ADD COLUMN integration_id BIGINT REFERENCES integrations(id) ON DELETE SET NULL;
|
|
||||||
|
|
||||||
CREATE INDEX alerts_integration_idx ON alerts (integration_id, received_at)
|
|
||||||
WHERE integration_id IS NOT NULL;
|
|
||||||
@@ -1,60 +0,0 @@
|
|||||||
-- Single sign-on through an OpenID Connect provider (Authentik, and anything
|
|
||||||
-- else that speaks OIDC).
|
|
||||||
--
|
|
||||||
-- Four things change, and none of them touches a password user: every new column
|
|
||||||
-- has a default that says "this is how it has always worked".
|
|
||||||
--
|
|
||||||
-- 1. user_identities says which provider account a user is. It is keyed on
|
|
||||||
-- (issuer, subject), never on email or username: those are mutable at the
|
|
||||||
-- provider, and a recycled address must not inherit somebody's account. A
|
|
||||||
-- user can have several identities (a second provider later), and none at all
|
|
||||||
-- (a local, password-only user), which is why this is a table and not two
|
|
||||||
-- columns on users.
|
|
||||||
--
|
|
||||||
-- 2. team_members.source and users.admin_source record who granted a role. 'oidc'
|
|
||||||
-- rows are owned by the group sync: it adds them when a group grants access
|
|
||||||
-- and removes them when it stops, and nothing else may edit them. 'manual' rows
|
|
||||||
-- are everything that existed before this migration, and are never touched by
|
|
||||||
-- the sync. Without the marker the sync could not tell a membership it created
|
|
||||||
-- from one an owner added by hand, and would have to either leave stale access
|
|
||||||
-- behind or delete people it had no business deleting.
|
|
||||||
--
|
|
||||||
-- 3. sessions.max_expires_at is a hard ceiling on a session's life. Ordinary
|
|
||||||
-- sessions slide for as long as they are used; a session made by an SSO login
|
|
||||||
-- must not, because the login is the only moment the groups are re-read.
|
|
||||||
-- Capping the session is what makes "removed from the group in the provider"
|
|
||||||
-- take effect within a bounded time. NULL means no ceiling.
|
|
||||||
--
|
|
||||||
-- 4. oidc_logins holds a login that has been started and not yet finished: the
|
|
||||||
-- state, nonce and PKCE verifier the callback must see again. A row rather
|
|
||||||
-- than a signed cookie, so it survives a restart and needs no signing key.
|
|
||||||
-- Only the hash of the state is stored, like every other token here; the
|
|
||||||
-- nonce and verifier are useless without the state that names the row.
|
|
||||||
CREATE TABLE user_identities (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
issuer TEXT NOT NULL,
|
|
||||||
subject TEXT NOT NULL,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
last_login_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
UNIQUE (issuer, subject)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX user_identities_user_idx ON user_identities (user_id);
|
|
||||||
|
|
||||||
ALTER TABLE team_members
|
|
||||||
ADD COLUMN source TEXT NOT NULL DEFAULT 'manual' CHECK (source IN ('manual', 'oidc'));
|
|
||||||
|
|
||||||
ALTER TABLE users
|
|
||||||
ADD COLUMN admin_source TEXT NOT NULL DEFAULT 'manual' CHECK (admin_source IN ('manual', 'oidc'));
|
|
||||||
|
|
||||||
ALTER TABLE sessions ADD COLUMN max_expires_at BIGINT;
|
|
||||||
|
|
||||||
CREATE TABLE oidc_logins (
|
|
||||||
state_hash TEXT PRIMARY KEY,
|
|
||||||
nonce TEXT NOT NULL,
|
|
||||||
pkce_verifier TEXT NOT NULL,
|
|
||||||
expires_at BIGINT NOT NULL
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX oidc_logins_expires_idx ON oidc_logins (expires_at);
|
|
||||||
@@ -1,40 +0,0 @@
|
|||||||
-- Signing in from a terminal, for clients that cannot open a browser on the
|
|
||||||
-- machine they run on (the TUI over SSH is the reason).
|
|
||||||
--
|
|
||||||
-- The flow is the OAuth device authorization grant, run by terdut itself rather
|
|
||||||
-- than the identity provider, so the terminal never talks to the provider and
|
|
||||||
-- the server issues its ordinary session at the end:
|
|
||||||
--
|
|
||||||
-- 1. The terminal asks for a login and gets two secrets: a device code it
|
|
||||||
-- keeps and polls with, and a short user code it shows the person.
|
|
||||||
-- 2. The person opens the verification URL on any device, signs in by whatever
|
|
||||||
-- means the server offers, sees the user code, and approves it.
|
|
||||||
-- 3. The terminal's next poll finds the row approved and is given a session.
|
|
||||||
--
|
|
||||||
-- Only the hash of the device code is stored, like every other token here: the
|
|
||||||
-- device code is what earns a session, so a database read must not yield one.
|
|
||||||
-- The user code is shown on screens and typed by people, so it is stored as is;
|
|
||||||
-- on its own it can only be approved, never redeemed.
|
|
||||||
--
|
|
||||||
-- user_id is the person who approved. It is empty until then, and the session
|
|
||||||
-- is minted at redemption, not at approval: an approval nobody collects must not
|
|
||||||
-- leave a live session lying about.
|
|
||||||
--
|
|
||||||
-- last_polled_at lets the server refuse a client that polls faster than the
|
|
||||||
-- interval it was told.
|
|
||||||
CREATE TABLE device_logins (
|
|
||||||
device_hash TEXT PRIMARY KEY,
|
|
||||||
user_code TEXT NOT NULL UNIQUE,
|
|
||||||
status TEXT NOT NULL DEFAULT 'pending' CHECK (status IN ('pending', 'approved', 'denied')),
|
|
||||||
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
expires_at BIGINT NOT NULL,
|
|
||||||
last_polled_at BIGINT NOT NULL DEFAULT 0
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX device_logins_expires_idx ON device_logins (expires_at);
|
|
||||||
|
|
||||||
-- Where to send the browser once a single sign-on login completes. A person who
|
|
||||||
-- opens /device?code=... without a session has to sign in first and then come
|
|
||||||
-- back to it, and the same is true of any other deep link. Validated when it is
|
|
||||||
-- stored: only a path on this server is ever kept.
|
|
||||||
ALTER TABLE oidc_logins ADD COLUMN next TEXT NOT NULL DEFAULT '/';
|
|
||||||
@@ -1,26 +0,0 @@
|
|||||||
-- Per-team OIDC group configuration, replacing the global
|
|
||||||
-- TERDUT_OIDC_GROUP_MAPPINGS env var.
|
|
||||||
--
|
|
||||||
-- Group -> team -> role used to be one global list an operator set for the
|
|
||||||
-- whole install, matched against a team by name, and the sync would create
|
|
||||||
-- the team if no team by that name existed yet. That put the decision of
|
|
||||||
-- which group controls a team in the server's environment rather than the
|
|
||||||
-- team's own hands, meant changing it needed an env var edit and a restart,
|
|
||||||
-- and let a typo in a team name silently create a stray team.
|
|
||||||
--
|
|
||||||
-- Each team now names, itself, which group grants membership and which
|
|
||||||
-- grants ownership. Nullable: most teams need neither. No uniqueness
|
|
||||||
-- constraint on either column — two teams may legitimately watch the same
|
|
||||||
-- provider group (a broad team and a narrower one both keyed off overlapping
|
|
||||||
-- groups is a choice for their owners to make, not one the schema should
|
|
||||||
-- refuse).
|
|
||||||
--
|
|
||||||
-- BREAKING CHANGE, deliberately not auto-migrated: TERDUT_OIDC_GROUP_MAPPINGS
|
|
||||||
-- stops being read as of this version, and the sync no longer creates a team
|
|
||||||
-- by name. Every team's group binding must be set again through
|
|
||||||
-- PUT /api/teams/{teamID}/oidc-groups. Until an owner does that, an
|
|
||||||
-- OIDC-sourced membership in that team is dropped at that user's next SSO
|
|
||||||
-- sign-in, the same way any other loss of group access is handled. See the
|
|
||||||
-- README's OIDC section.
|
|
||||||
ALTER TABLE teams ADD COLUMN oidc_member_group TEXT;
|
|
||||||
ALTER TABLE teams ADD COLUMN oidc_owner_group TEXT;
|
|
||||||
@@ -1,43 +0,0 @@
|
|||||||
-- Service accounts: a scoped, non-human credential for automation (e.g.
|
|
||||||
-- terdut-operator) that needs to manage teams, escalation policies, dead
|
|
||||||
-- man's switches, integrations and OIDC group bindings without impersonating
|
|
||||||
-- a human user. See SERVICE-ACCOUNTS.md for the design this implements.
|
|
||||||
--
|
|
||||||
-- Deliberately not a users row: no password_hash, no is_admin, no
|
|
||||||
-- user_identities linkage, so a service account can never be pulled into
|
|
||||||
-- OIDC group sync or password login, and is never mistaken for a human in an
|
|
||||||
-- audit trail.
|
|
||||||
--
|
|
||||||
-- scope is 'instance' (acts with the same reach system administration has
|
|
||||||
-- over teams: create one, list them, mint a 'team'-scoped account against
|
|
||||||
-- any of them) or 'team' (acts as that one team's owner, and nothing else).
|
|
||||||
-- The CHECK ties team_id's presence to scope directly, rather than leaving it
|
|
||||||
-- to application code to keep the two consistent.
|
|
||||||
CREATE TABLE service_accounts (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
name TEXT NOT NULL UNIQUE,
|
|
||||||
scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
|
|
||||||
team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
CONSTRAINT service_accounts_scope_team_id_chk CHECK (
|
|
||||||
(scope = 'team' AND team_id IS NOT NULL) OR
|
|
||||||
(scope = 'instance' AND team_id IS NULL)
|
|
||||||
)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX service_accounts_team_id_idx ON service_accounts(team_id);
|
|
||||||
|
|
||||||
-- One account, many keys: rotation is minting a new one and revoking the
|
|
||||||
-- old, the same shape api_keys already has, so an account's identity and
|
|
||||||
-- audit history survive a rotation instead of being recreated by it.
|
|
||||||
CREATE TABLE service_account_keys (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
|
|
||||||
key_hash TEXT NOT NULL UNIQUE,
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
last_used_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys(service_account_id);
|
|
||||||
@@ -1,39 +0,0 @@
|
|||||||
-- Service-account actors on incident mutations (terdut-server#25). A
|
|
||||||
-- team-scoped service account acknowledging/resolving/snoozing/noting an
|
|
||||||
-- incident is not a users row, so it cannot be written into
|
|
||||||
-- acknowledged_by/incident_events.user_id — doing so either violates the
|
|
||||||
-- users(id) FK (new rows) or, for incident_events.user_id, silently matches
|
|
||||||
-- zero rows on delete. These columns are the service-account-shaped parallel
|
|
||||||
-- to the existing human ones: nullable, mutually exclusive with their human
|
|
||||||
-- counterpart, ON DELETE SET NULL so a deleted service account doesn't take
|
|
||||||
-- the incident history with it.
|
|
||||||
ALTER TABLE incidents
|
|
||||||
ADD COLUMN acknowledged_by_service_account_id BIGINT
|
|
||||||
REFERENCES service_accounts(id) ON DELETE SET NULL;
|
|
||||||
|
|
||||||
ALTER TABLE incident_events
|
|
||||||
ADD COLUMN service_account_id BIGINT
|
|
||||||
REFERENCES service_accounts(id) ON DELETE SET NULL;
|
|
||||||
|
|
||||||
-- At most one actor kind per row: both NULL ("the server acted") is valid,
|
|
||||||
-- exactly one set is valid, both set is a bug this constraint refuses to
|
|
||||||
-- store rather than silently accepting.
|
|
||||||
ALTER TABLE incidents
|
|
||||||
ADD CONSTRAINT incidents_ack_actor_xor_chk CHECK (
|
|
||||||
acknowledged_by IS NULL OR acknowledged_by_service_account_id IS NULL
|
|
||||||
);
|
|
||||||
|
|
||||||
ALTER TABLE incident_events
|
|
||||||
ADD CONSTRAINT incident_events_actor_xor_chk CHECK (
|
|
||||||
user_id IS NULL OR service_account_id IS NULL
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incidents_acknowledged_by_service_account_id_idx
|
|
||||||
ON incidents(acknowledged_by_service_account_id);
|
|
||||||
CREATE INDEX incident_events_service_account_id_idx
|
|
||||||
ON incident_events(service_account_id);
|
|
||||||
|
|
||||||
-- assigned_to_service_account_id is deliberately not added here: it would sit
|
|
||||||
-- unpopulated until handleIncidentAssign itself tracks an actor, which is a
|
|
||||||
-- separate, pre-existing gap (it records the assignee today, never the
|
|
||||||
-- actor, for humans either) tracked in its own follow-up issue.
|
|
||||||
@@ -1,16 +0,0 @@
|
|||||||
-- Backs the rate limiters (failed logins, sign-ups, OIDC/device start) with
|
|
||||||
-- Postgres instead of an in-memory map, now that the server runs more than
|
|
||||||
-- one replica in production (v0.37.0): a counter that only ever sees its own
|
|
||||||
-- pod's traffic quietly let every one of these limits through multiplied by
|
|
||||||
-- the replica count.
|
|
||||||
--
|
|
||||||
-- window_start is the start of the current fixed window for key, in the same
|
|
||||||
-- "unix seconds" shape every other timestamp in this schema uses. The window
|
|
||||||
-- resets rather than slides, matching the in-memory limiter it replaces:
|
|
||||||
-- once a key's window is older than the limiter's window length, the next
|
|
||||||
-- failure starts a fresh one instead of extending the stale one.
|
|
||||||
CREATE TABLE rate_limit_counters (
|
|
||||||
key TEXT PRIMARY KEY,
|
|
||||||
window_start BIGINT NOT NULL,
|
|
||||||
count INT NOT NULL
|
|
||||||
);
|
|
||||||
@@ -1,7 +0,0 @@
|
|||||||
-- Optional expiry on a user's own API keys. NULL (the existing default for
|
|
||||||
-- every row already in this table) means "never expires" -- the same
|
|
||||||
-- behavior these keys have always had, so no existing integration breaks.
|
|
||||||
-- Service account keys are deliberately NOT touched: they are a different
|
|
||||||
-- table, managed by automation, and already distinguished by their own
|
|
||||||
-- "tdsa_" prefix.
|
|
||||||
ALTER TABLE api_keys ADD COLUMN expires_at BIGINT;
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
-- Who performed an assignment (terdut-server#35). On an 'assigned' event
|
|
||||||
-- incident_events.user_id is the assignee, so the actor needs columns of its
|
|
||||||
-- own. Only populated for 'assigned' events; every other event type keeps
|
|
||||||
-- using user_id/service_account_id for the actor. Older 'assigned' rows stay
|
|
||||||
-- NULL (the actor was never recorded). Same shape as migration 015: nullable,
|
|
||||||
-- mutually exclusive, ON DELETE SET NULL.
|
|
||||||
--
|
|
||||||
-- assigned_to_service_account_id is still deliberately not added: making
|
|
||||||
-- service accounts assignable is a separate change (request body, assignee
|
|
||||||
-- picker, notifier, filters).
|
|
||||||
ALTER TABLE incident_events
|
|
||||||
ADD COLUMN actor_user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
ADD COLUMN actor_service_account_id BIGINT REFERENCES service_accounts(id) ON DELETE SET NULL;
|
|
||||||
|
|
||||||
ALTER TABLE incident_events
|
|
||||||
ADD CONSTRAINT incident_events_assign_actor_xor_chk CHECK (
|
|
||||||
actor_user_id IS NULL OR actor_service_account_id IS NULL
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incident_events_actor_user_id_idx ON incident_events(actor_user_id);
|
|
||||||
CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events(actor_service_account_id);
|
|
||||||
@@ -33,7 +33,7 @@ type Alert struct {
|
|||||||
// refreshed. The sweeper stale-dates against it (see expireStale), API
|
// refreshed. The sweeper stale-dates against it (see expireStale), API
|
||||||
// clients render it, and GET /api/alerts is ordered by it. Anything that
|
// clients render it, and GET /api/alerts is ordered by it. Anything that
|
||||||
// stops the webhook handler from advancing it on a re-send is a breaking
|
// stops the webhook handler from advancing it on a re-send is a breaking
|
||||||
// change — see "received_at is a liveness heartbeat" in the README and
|
// change — see "received_at is a liveness heartbeat" in docs/api.md and
|
||||||
// TestWebhook_ResendBumpsReceivedAt.
|
// TestWebhook_ResendBumpsReceivedAt.
|
||||||
ReceivedAt time.Time `json:"received_at"`
|
ReceivedAt time.Time `json:"received_at"`
|
||||||
|
|
||||||
@@ -52,7 +52,7 @@ type Alert struct {
|
|||||||
// inferred. Under "expiry" nothing ever reported an end, so EndsAt is only
|
// inferred. Under "expiry" nothing ever reported an end, so EndsAt is only
|
||||||
// an upper bound (see expireStale) and ReceivedAt is the more truthful
|
// an upper bound (see expireStale) and ReceivedAt is the more truthful
|
||||||
// signal. Treat the value set as open — see "resolution_source says how much
|
// signal. Treat the value set as open — see "resolution_source says how much
|
||||||
// to trust ends_at" in the README, and TestWebhook_ResolvedSetsSource /
|
// to trust ends_at" in docs/api.md, and TestWebhook_ResolvedSetsSource /
|
||||||
// TestExpiry_StaleFiringAlert.
|
// TestExpiry_StaleFiringAlert.
|
||||||
ResolutionSource *string `json:"resolution_source,omitempty"`
|
ResolutionSource *string `json:"resolution_source,omitempty"`
|
||||||
|
|
||||||
|
|||||||
@@ -9,6 +9,10 @@ type Team struct {
|
|||||||
Name string `json:"name"`
|
Name string `json:"name"`
|
||||||
CreatedAt time.Time `json:"created_at"`
|
CreatedAt time.Time `json:"created_at"`
|
||||||
|
|
||||||
|
// ExternalID identifies a team managed by automation; see handleCreateTeam.
|
||||||
|
// Shown to instance service accounts and admins only.
|
||||||
|
ExternalID *string `json:"external_id,omitempty"`
|
||||||
|
|
||||||
// Role is the caller's own role in this team, populated when a team is
|
// Role is the caller's own role in this team, populated when a team is
|
||||||
// listed for a particular person. Empty when nobody in particular is
|
// listed for a particular person. Empty when nobody in particular is
|
||||||
// asking, as in the admin listing.
|
// asking, as in the admin listing.
|
||||||
|
|||||||
@@ -102,6 +102,3 @@ func rank(role string) int {
|
|||||||
}
|
}
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
// HigherRole reports whether role a outranks role b.
|
|
||||||
func HigherRole(a, b string) bool { return rank(a) > rank(b) }
|
|
||||||
|
|||||||
@@ -296,6 +296,7 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
|||||||
border-top: 1px solid var(--border);
|
border-top: 1px solid var(--border);
|
||||||
}
|
}
|
||||||
.nav-brand { display: none; }
|
.nav-brand { display: none; }
|
||||||
|
.nav-sep, .nav-avatar { display: none; }
|
||||||
/* Hidden here (shown from 900px below): on the phone bar the team switcher
|
/* Hidden here (shown from 900px below): on the phone bar the team switcher
|
||||||
lives in the topbar instead, as #team-selector-mobile. */
|
lives in the topbar instead, as #team-selector-mobile. */
|
||||||
.nav-team-selector { display: none; }
|
.nav-team-selector { display: none; }
|
||||||
@@ -305,7 +306,10 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
|||||||
block below cancels it back to a natural-width row item. */
|
block below cancels it back to a natural-width row item. */
|
||||||
flex: 1 1 0;
|
flex: 1 1 0;
|
||||||
display: flex; flex-direction: column; align-items: center; justify-content: center; gap: 2px;
|
display: flex; flex-direction: column; align-items: center; justify-content: center; gap: 2px;
|
||||||
color: var(--faint); font-size: 11px; font-weight: 600;
|
color: var(--muted); font-size: 12px; font-weight: 600;
|
||||||
|
/* The More tab is a <button>; without this it keeps the browser's grey box
|
||||||
|
and reads as a highlighted tab beside four plain links. */
|
||||||
|
background: none; border: 0; font-family: inherit; cursor: pointer;
|
||||||
/* min-width lets a column shrink below its label's natural width, which is
|
/* min-width lets a column shrink below its label's natural width, which is
|
||||||
what stops six tabs widening the bar past the screen. */
|
what stops six tabs widening the bar past the screen. */
|
||||||
min-width: 0; padding: 0 2px;
|
min-width: 0; padding: 0 2px;
|
||||||
@@ -320,6 +324,8 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
|||||||
.nav-link svg { width: 24px; height: 24px; flex: none; fill: none; stroke: currentColor; stroke-width: 1.8; stroke-linecap: round; stroke-linejoin: round; }
|
.nav-link svg { width: 24px; height: 24px; flex: none; fill: none; stroke: currentColor; stroke-width: 1.8; stroke-linecap: round; stroke-linejoin: round; }
|
||||||
|
|
||||||
.nav-link[aria-current="page"] { color: var(--accent); }
|
.nav-link[aria-current="page"] { color: var(--accent); }
|
||||||
|
/* The open tab's icon is filled, so it is not told apart by colour alone. */
|
||||||
|
.nav-link[aria-current="page"] svg { fill: currentColor; fill-opacity: 0.16; }
|
||||||
.nav-badge {
|
.nav-badge {
|
||||||
position: absolute; top: 6px; left: calc(50% + 6px);
|
position: absolute; top: 6px; left: calc(50% + 6px);
|
||||||
min-width: 18px; height: 18px; padding: 0 5px;
|
min-width: 18px; height: 18px; padding: 0 5px;
|
||||||
@@ -389,16 +395,23 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
|||||||
.chip .count { margin-left: 4px; opacity: 0.7; }
|
.chip .count { margin-left: 4px; opacity: 0.7; }
|
||||||
/* Nothing in it: step back so the chips that have something stand out. */
|
/* Nothing in it: step back so the chips that have something stand out. */
|
||||||
.chip .count.zero { opacity: 0.4; }
|
.chip .count.zero { opacity: 0.4; }
|
||||||
/* An overlay, not a flex item: absolute against .chips' own (non-scrolling)
|
/* A scrolling strip fades on the right edge while there is more to scroll to
|
||||||
box stays flush with its real right edge regardless of scroll position,
|
(fadeOnOverflow in ui.js sets data-more). A mask on the strip itself, not an
|
||||||
which turned out not to be true of position:sticky here — as a flex
|
element inside it: anything inside a scroller scrolls away with the content,
|
||||||
item, its sticky offset interacted with the row's gap and its own
|
which is what the old overlay did. */
|
||||||
negative margin, landing short of the edge by about one gap's width. */
|
[data-more] {
|
||||||
.chips-fade {
|
-webkit-mask-image: linear-gradient(90deg, #000 calc(100% - 28px), transparent);
|
||||||
position: absolute; top: 0; right: 0; bottom: 0;
|
mask-image: linear-gradient(90deg, #000 calc(100% - 28px), transparent);
|
||||||
width: 24px;
|
}
|
||||||
background: linear-gradient(to right, transparent, var(--bg));
|
|
||||||
pointer-events: none;
|
/* The queue's cluster filter: a pill-shaped select under the status chips. */
|
||||||
|
.queue-origin { position: relative; display: flex; align-items: center; padding: 0 16px 8px; }
|
||||||
|
.origin-select-icon { position: absolute; left: 28px; width: 16px; height: 16px; color: var(--muted); pointer-events: none; }
|
||||||
|
.origin-select {
|
||||||
|
min-height: 34px; max-width: 100%; padding: 0 12px 0 34px;
|
||||||
|
border: 1px solid var(--border-strong); border-radius: 999px;
|
||||||
|
background: var(--surface); color: var(--text);
|
||||||
|
font: inherit; font-size: 13px; font-weight: 600; cursor: pointer;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ---------- lists ---------- */
|
/* ---------- lists ---------- */
|
||||||
@@ -440,7 +453,6 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
|||||||
/* Which team's queue a row came from. Only rendered for somebody in more than
|
/* Which team's queue a row came from. Only rendered for somebody in more than
|
||||||
one team, so it never repeats the same word down the whole list. */
|
one team, so it never repeats the same word down the whole list. */
|
||||||
/* Separates the status chips from the team chips in the queue's filter row. */
|
/* Separates the status chips from the team chips in the queue's filter row. */
|
||||||
.chip-sep { width: 1px; align-self: stretch; background: var(--border); margin: 0 2px; }
|
|
||||||
|
|
||||||
.row-team {
|
.row-team {
|
||||||
padding: 1px 6px; border-radius: 4px;
|
padding: 1px 6px; border-radius: 4px;
|
||||||
@@ -486,7 +498,15 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
|||||||
}
|
}
|
||||||
.badge::before { content: ""; width: 7px; height: 7px; border-radius: 50%; background: currentColor; }
|
.badge::before { content: ""; width: 7px; height: 7px; border-radius: 50%; background: currentColor; }
|
||||||
.badge.plain::before { display: none; }
|
.badge.plain::before { display: none; }
|
||||||
.badge .badge-icon { width: 12px; height: 12px; stroke-width: 2.4; }
|
.badge .badge-icon, .origin-chip .badge-icon { width: 12px; height: 12px; stroke-width: 2.4; }
|
||||||
|
/* The cluster an incident or alert came from. Same shape as a badge, coloured
|
||||||
|
from the rcN palette (see originClass in format.js), never the severity one. */
|
||||||
|
.origin-chip {
|
||||||
|
display: inline-flex; align-items: center; gap: 5px; max-width: 100%;
|
||||||
|
padding: 1px 8px; border-radius: 999px;
|
||||||
|
font-size: 12px; font-weight: 700; letter-spacing: 0.01em; white-space: nowrap;
|
||||||
|
overflow: hidden; text-overflow: ellipsis;
|
||||||
|
}
|
||||||
.badge.st-triggered, .badge.st-firing { background: var(--crit-soft); color: var(--crit); }
|
.badge.st-triggered, .badge.st-firing { background: var(--crit-soft); color: var(--crit); }
|
||||||
.badge.st-acknowledged { background: var(--warn-soft); color: var(--warn); }
|
.badge.st-acknowledged { background: var(--warn-soft); color: var(--warn); }
|
||||||
.badge.st-snoozed { background: var(--snooze-soft); color: var(--snooze); }
|
.badge.st-snoozed { background: var(--snooze-soft); color: var(--snooze); }
|
||||||
@@ -564,11 +584,25 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
|||||||
.alert-item-summary { color: var(--muted); font-size: 14px; overflow-wrap: anywhere; }
|
.alert-item-summary { color: var(--muted); font-size: 14px; overflow-wrap: anywhere; }
|
||||||
.alert-item-foot { display: flex; flex-wrap: wrap; gap: 4px 12px; font-size: 13px; color: var(--faint); }
|
.alert-item-foot { display: flex; flex-wrap: wrap; gap: 4px 12px; font-size: 13px; color: var(--faint); }
|
||||||
.alert-item-foot a { color: var(--accent); font-weight: 600; }
|
.alert-item-foot a { color: var(--accent); font-weight: 600; }
|
||||||
details > summary { cursor: pointer; color: var(--muted); font-size: 13px; font-weight: 600; list-style: none; }
|
/* A disclosure is a button: a chevron that turns, and a 44px target on a
|
||||||
|
touch screen (the bare triangle it replaces was a few pixels wide). */
|
||||||
|
details > summary {
|
||||||
|
display: flex; align-items: center; gap: 8px;
|
||||||
|
min-height: 44px; margin: 0 -6px; padding: 0 6px; border-radius: var(--radius-sm);
|
||||||
|
cursor: pointer; color: var(--muted); font-size: 14px; font-weight: 600; list-style: none;
|
||||||
|
}
|
||||||
details > summary::-webkit-details-marker { display: none; }
|
details > summary::-webkit-details-marker { display: none; }
|
||||||
details > summary::before { content: "▸ "; }
|
details > summary::before {
|
||||||
details[open] > summary::before { content: "▾ "; }
|
content: ""; flex: none; width: 7px; height: 7px; margin: 0 5px 0 3px;
|
||||||
|
border-right: 2px solid currentColor; border-bottom: 2px solid currentColor;
|
||||||
|
transform: rotate(-45deg); transition: transform 0.12s;
|
||||||
|
}
|
||||||
|
details[open] > summary::before { transform: rotate(45deg); }
|
||||||
|
details > summary:hover { background: var(--surface-2); color: var(--text); }
|
||||||
|
details > summary:focus-visible { outline: 2px solid var(--accent); outline-offset: 1px; }
|
||||||
details[open] > summary { margin-bottom: 8px; }
|
details[open] > summary { margin-bottom: 8px; }
|
||||||
|
@media (hover: hover) and (pointer: fine) { details > summary { min-height: 32px; } }
|
||||||
|
@media (prefers-reduced-motion: reduce) { details > summary::before { transition: none; } }
|
||||||
|
|
||||||
/* One .tl-phase per status the incident has been through (see
|
/* One .tl-phase per status the incident has been through (see
|
||||||
timelinePhases() in incident.js) — each with its own .timeline <ol>, so
|
timelinePhases() in incident.js) — each with its own .timeline <ol>, so
|
||||||
@@ -827,6 +861,17 @@ kbd {
|
|||||||
.nav-team-selector { display: inline-flex; margin: -8px 10px 14px; width: calc(100% - 20px); }
|
.nav-team-selector { display: inline-flex; margin: -8px 10px 14px; width: calc(100% - 20px); }
|
||||||
.nav-link-secondary { display: flex; }
|
.nav-link-secondary { display: flex; }
|
||||||
.nav-more-btn { display: none; }
|
.nav-more-btn { display: none; }
|
||||||
|
.nav-sep { display: block; height: 1px; margin: 8px 10px; background: var(--border); }
|
||||||
|
/* The account link sits at the foot of the sidebar, as the signed-in person. */
|
||||||
|
.nav-sep-account { margin-top: auto; }
|
||||||
|
.nav-account .nav-avatar {
|
||||||
|
display: grid; place-items: center; flex: none;
|
||||||
|
width: 28px; height: 28px; margin: -4px 0 -4px -4px; border-radius: 50%;
|
||||||
|
background: var(--accent-soft); color: var(--accent);
|
||||||
|
font-size: 13px; font-weight: 750; text-transform: uppercase;
|
||||||
|
}
|
||||||
|
.nav-account .nav-avatar:not([hidden]) ~ .nav-label { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||||
|
.nav-account svg:has(~ .nav-avatar:not([hidden])) { display: none; }
|
||||||
.nav-link {
|
.nav-link {
|
||||||
flex: none; flex-direction: row; justify-content: flex-start; gap: 12px;
|
flex: none; flex-direction: row; justify-content: flex-start; gap: 12px;
|
||||||
min-height: 40px; padding: 0 10px; border-radius: var(--radius-sm);
|
min-height: 40px; padding: 0 10px; border-radius: var(--radius-sm);
|
||||||
@@ -834,6 +879,11 @@ kbd {
|
|||||||
}
|
}
|
||||||
.nav-link:hover { background: var(--surface-2); }
|
.nav-link:hover { background: var(--surface-2); }
|
||||||
.nav-link[aria-current="page"] { background: var(--accent-soft); color: var(--accent); }
|
.nav-link[aria-current="page"] { background: var(--accent-soft); color: var(--accent); }
|
||||||
|
/* A 3px bar on the active item, as well as the tint. */
|
||||||
|
.nav-link[aria-current="page"]::before {
|
||||||
|
content: ""; position: absolute; left: -12px; top: 8px; bottom: 8px; width: 3px;
|
||||||
|
border-radius: 0 3px 3px 0; background: var(--accent);
|
||||||
|
}
|
||||||
.nav-link svg { width: 20px; height: 20px; }
|
.nav-link svg { width: 20px; height: 20px; }
|
||||||
.nav-badge { position: static; margin-left: auto; }
|
.nav-badge { position: static; margin-left: auto; }
|
||||||
|
|
||||||
@@ -851,8 +901,6 @@ kbd {
|
|||||||
/* The pane is 340-420px wide and a mouse cannot scroll a row whose scrollbar
|
/* The pane is 340-420px wide and a mouse cannot scroll a row whose scrollbar
|
||||||
is hidden, so the chips wrap here instead: Archived stays reachable. */
|
is hidden, so the chips wrap here instead: Archived stays reachable. */
|
||||||
.pane-list .chips { flex-wrap: wrap; overflow-x: visible; }
|
.pane-list .chips { flex-wrap: wrap; overflow-x: visible; }
|
||||||
.pane-list .chip-sep { display: none; }
|
|
||||||
.chips-fade { display: none; }
|
|
||||||
/* With nothing selected there is no detail to show next to, so the list
|
/* With nothing selected there is no detail to show next to, so the list
|
||||||
takes the whole row instead of leaving the second column as dead space
|
takes the whole row instead of leaving the second column as dead space
|
||||||
around the placeholder text. Selecting an incident (.has-detail) drops
|
around the placeholder text. Selecting an incident (.has-detail) drops
|
||||||
@@ -901,9 +949,12 @@ kbd {
|
|||||||
.admin-table th {
|
.admin-table th {
|
||||||
text-align: left; font-weight: 600; color: var(--muted); font-size: 12px;
|
text-align: left; font-weight: 600; color: var(--muted); font-size: 12px;
|
||||||
text-transform: uppercase; letter-spacing: 0.04em;
|
text-transform: uppercase; letter-spacing: 0.04em;
|
||||||
padding: 4px 8px 4px 0; border-bottom: 1px solid var(--border);
|
padding: 4px 16px 4px 0; border-bottom: 1px solid var(--border);
|
||||||
}
|
}
|
||||||
.admin-table td { padding: 8px 8px 8px 0; border-bottom: 1px solid var(--border); vertical-align: middle; }
|
/* 16px between columns, so a right-aligned count never touches the
|
||||||
|
left-aligned text beside it ("5" against "16d ago"). */
|
||||||
|
.admin-table td { padding: 10px 16px 10px 0; border-bottom: 1px solid var(--border); vertical-align: middle; }
|
||||||
|
.admin-table th:last-child, .admin-table td:last-child { padding-right: 0; }
|
||||||
.admin-table tr:last-child td { border-bottom: none; }
|
.admin-table tr:last-child td { border-bottom: none; }
|
||||||
.admin-table .num { text-align: right; font-variant-numeric: tabular-nums; }
|
.admin-table .num { text-align: right; font-variant-numeric: tabular-nums; }
|
||||||
.admin-table td .btn-sm + .btn-sm { margin-left: 6px; }
|
.admin-table td .btn-sm + .btn-sm { margin-left: 6px; }
|
||||||
@@ -1199,11 +1250,6 @@ button.rota-week:hover { background: var(--surface-2); color: var(--text); }
|
|||||||
overflow-x: auto; scrollbar-width: none;
|
overflow-x: auto; scrollbar-width: none;
|
||||||
}
|
}
|
||||||
.subnav::-webkit-scrollbar { display: none; }
|
.subnav::-webkit-scrollbar { display: none; }
|
||||||
/* A fade on the right edge says "there is more" where the strip overflows;
|
|
||||||
the desktop width fits every entry, so it is phone-only. */
|
|
||||||
@media (max-width: 899px) {
|
|
||||||
.subnav { -webkit-mask-image: linear-gradient(90deg, #000 calc(100% - 28px), transparent); mask-image: linear-gradient(90deg, #000 calc(100% - 28px), transparent); }
|
|
||||||
}
|
|
||||||
.subnav-link {
|
.subnav-link {
|
||||||
flex: none;
|
flex: none;
|
||||||
padding: 8px 12px; margin-bottom: -1px;
|
padding: 8px 12px; margin-bottom: -1px;
|
||||||
|
|||||||
@@ -110,6 +110,9 @@
|
|||||||
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M4 20h16M7 20v-7M12 20V6M17 20v-10"/></svg>
|
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M4 20h16M7 20v-7M12 20V6M17 20v-10"/></svg>
|
||||||
<span class="nav-label">Stats</span>
|
<span class="nav-label">Stats</span>
|
||||||
</a>
|
</a>
|
||||||
|
<!-- Dividers between the groups (Queue, On-call, Alerts, Stats | Team,
|
||||||
|
Admin | Account). Desktop sidebar only; the phone bar has no room. -->
|
||||||
|
<span class="nav-sep" aria-hidden="true"></span>
|
||||||
<a class="nav-link" href="/team" data-section="team" aria-label="Team">
|
<a class="nav-link" href="/team" data-section="team" aria-label="Team">
|
||||||
<svg viewBox="0 0 24 24" aria-hidden="true"><circle cx="9" cy="8" r="3"/><circle cx="17" cy="9" r="2.5"/><path d="M3 19a6 6 0 0 1 12 0M15 19a5 5 0 0 1 6-4"/></svg>
|
<svg viewBox="0 0 24 24" aria-hidden="true"><circle cx="9" cy="8" r="3"/><circle cx="17" cy="9" r="2.5"/><path d="M3 19a6 6 0 0 1 12 0M15 19a5 5 0 0 1 6-4"/></svg>
|
||||||
<span class="nav-label">Team</span>
|
<span class="nav-label">Team</span>
|
||||||
@@ -121,9 +124,14 @@
|
|||||||
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M12 3l7 3v6c0 4-3 7-7 9-4-2-7-5-7-9V6z"/></svg>
|
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M12 3l7 3v6c0 4-3 7-7 9-4-2-7-5-7-9V6z"/></svg>
|
||||||
<span class="nav-label">Admin</span>
|
<span class="nav-label">Admin</span>
|
||||||
</a>
|
</a>
|
||||||
<a class="nav-link nav-link-secondary" href="/more" data-section="more" aria-label="Account">
|
<span class="nav-sep nav-sep-account" aria-hidden="true"></span>
|
||||||
|
<!-- At the foot of the desktop sidebar, as the signed-in person: app.js
|
||||||
|
fills the avatar and the name from /api/me, and until then it reads
|
||||||
|
"Account". -->
|
||||||
|
<a class="nav-link nav-link-secondary nav-account" href="/more" data-section="more" aria-label="Account">
|
||||||
<svg viewBox="0 0 24 24" aria-hidden="true"><circle cx="12" cy="8" r="3.5"/><path d="M5 20a7 7 0 0 1 14 0"/></svg>
|
<svg viewBox="0 0 24 24" aria-hidden="true"><circle cx="12" cy="8" r="3.5"/><path d="M5 20a7 7 0 0 1 14 0"/></svg>
|
||||||
<span class="nav-label">Account</span>
|
<span class="nav-avatar" id="nav-avatar" aria-hidden="true" hidden></span>
|
||||||
|
<span class="nav-label" id="nav-account-label">Account</span>
|
||||||
</a>
|
</a>
|
||||||
<!-- Phone-width only (see .nav-more-btn in app.css): opens the same
|
<!-- Phone-width only (see .nav-more-btn in app.css): opens the same
|
||||||
sheet the old hamburger button did, for the sections the bottom
|
sheet the old hamburger button did, for the sections the bottom
|
||||||
@@ -146,6 +154,9 @@
|
|||||||
<section id="view-queue" class="view view-queue" data-view="queue">
|
<section id="view-queue" class="view view-queue" data-view="queue">
|
||||||
<div class="pane pane-list">
|
<div class="pane pane-list">
|
||||||
<div class="chips" id="queue-filters" role="tablist" aria-label="Filter"></div>
|
<div class="chips" id="queue-filters" role="tablist" aria-label="Filter"></div>
|
||||||
|
<!-- The cluster filter. Hidden until the queue has seen two or more
|
||||||
|
clusters; queue.js fills it in. -->
|
||||||
|
<div class="queue-origin" id="queue-origin" hidden></div>
|
||||||
<div id="queue-list" class="list"></div>
|
<div id="queue-list" class="list"></div>
|
||||||
</div>
|
</div>
|
||||||
<div class="pane pane-detail" id="detail" aria-live="polite"></div>
|
<div class="pane pane-detail" id="detail" aria-live="polite"></div>
|
||||||
|
|||||||
@@ -13,7 +13,7 @@
|
|||||||
// gate.
|
// gate.
|
||||||
|
|
||||||
import * as api from './api.js';
|
import * as api from './api.js';
|
||||||
import { h, clear, spinner, confirm, menuCard, ssoBadge, SSO_MANAGED } from './ui.js';
|
import { h, clear, spinner, confirm, menuCard, ssoBadge, SSO_MANAGED, fadeOnOverflow } from './ui.js';
|
||||||
import { state, myID } from './state.js';
|
import { state, myID } from './state.js';
|
||||||
|
|
||||||
const view = () => document.getElementById('view-admin');
|
const view = () => document.getElementById('view-admin');
|
||||||
@@ -111,6 +111,7 @@ function subnav() {
|
|||||||
})));
|
})));
|
||||||
// On a phone the strip overflows; bring the open section into view so a
|
// On a phone the strip overflows; bring the open section into view so a
|
||||||
// tab past the edge (Sources, Switches) is never the one that is hidden.
|
// tab past the edge (Sources, Switches) is never the one that is hidden.
|
||||||
|
fadeOnOverflow(nav);
|
||||||
requestAnimationFrame(() => nav.querySelector('[aria-current]')
|
requestAnimationFrame(() => nav.querySelector('[aria-current]')
|
||||||
?.scrollIntoView({ inline: 'center', block: 'nearest' }));
|
?.scrollIntoView({ inline: 'center', block: 'nearest' }));
|
||||||
return nav;
|
return nav;
|
||||||
@@ -170,7 +171,7 @@ function teamsCard() {
|
|||||||
function newTeamForm() {
|
function newTeamForm() {
|
||||||
const name = h('input', { name: 'name', type: 'text', placeholder: 'New team name', required: true });
|
const name = h('input', { name: 'name', type: 'text', placeholder: 'New team name', required: true });
|
||||||
const form = h('form', { class: 'inline-form' }, name,
|
const form = h('form', { class: 'inline-form' }, name,
|
||||||
h('button', { class: 'btn', type: 'submit', text: 'Create' }));
|
h('button', { class: 'btn btn-primary', type: 'submit', text: 'Create' }));
|
||||||
form.addEventListener('submit', async (e) => {
|
form.addEventListener('submit', async (e) => {
|
||||||
e.preventDefault();
|
e.preventDefault();
|
||||||
if (busy) return;
|
if (busy) return;
|
||||||
|
|||||||
@@ -152,16 +152,19 @@ function identityCard() {
|
|||||||
// membership looks the way it does, but setting it is the team's own
|
// membership looks the way it does, but setting it is the team's own
|
||||||
// owner's call, from the Team tab.
|
// owner's call, from the Team tab.
|
||||||
...(state.auth?.oidc?.enabled ? [
|
...(state.auth?.oidc?.enabled ? [
|
||||||
fact('OIDC member group', t.oidc_member_group || '—'),
|
fact('OIDC member group', t.oidc_member_group || notConfigured()),
|
||||||
fact('OIDC owner group', t.oidc_owner_group || '—'),
|
fact('OIDC owner group', t.oidc_owner_group || notConfigured()),
|
||||||
] : []),
|
] : []),
|
||||||
),
|
),
|
||||||
form, err, ok,
|
form, err, ok,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const notConfigured = () => h('span', { class: 'muted', text: 'Not configured' });
|
||||||
|
|
||||||
|
// value is text or a node.
|
||||||
function fact(label, value) {
|
function fact(label, value) {
|
||||||
return [h('dt', { text: label }), h('dd', { text: value })];
|
return [h('dt', { text: label }), h('dd', {}, value)];
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- members ---------------------------------------------------------------
|
// --- members ---------------------------------------------------------------
|
||||||
@@ -207,7 +210,7 @@ function membersCard() {
|
|||||||
h('option', { value: 'member', text: 'member' }),
|
h('option', { value: 'member', text: 'member' }),
|
||||||
h('option', { value: 'owner', text: 'owner' }));
|
h('option', { value: 'owner', text: 'owner' }));
|
||||||
const form = h('form', { class: 'inline-form' }, pick, role,
|
const form = h('form', { class: 'inline-form' }, pick, role,
|
||||||
h('button', { class: 'btn', type: 'submit', text: 'Add' }));
|
h('button', { class: 'btn btn-primary', type: 'submit', text: 'Add' }));
|
||||||
form.addEventListener('submit', (e) => {
|
form.addEventListener('submit', (e) => {
|
||||||
e.preventDefault();
|
e.preventDefault();
|
||||||
act(() => api.addTeamMember(teamID, Number(pick.value), role.value));
|
act(() => api.addTeamMember(teamID, Number(pick.value), role.value));
|
||||||
@@ -238,7 +241,7 @@ function invitesCard() {
|
|||||||
h('option', { value: 'member', text: 'member' }),
|
h('option', { value: 'member', text: 'member' }),
|
||||||
h('option', { value: 'owner', text: 'owner' }));
|
h('option', { value: 'owner', text: 'owner' }));
|
||||||
const form = h('form', { class: 'inline-form' }, role,
|
const form = h('form', { class: 'inline-form' }, role,
|
||||||
h('button', { class: 'btn', type: 'submit', text: 'Create invite' }));
|
h('button', { class: 'btn btn-primary', type: 'submit', text: 'Create invite' }));
|
||||||
form.addEventListener('submit', async (e) => {
|
form.addEventListener('submit', async (e) => {
|
||||||
e.preventDefault();
|
e.preventDefault();
|
||||||
if (busy) return;
|
if (busy) return;
|
||||||
|
|||||||
@@ -197,7 +197,7 @@ function teamsCard() {
|
|||||||
h('option', { value: 'member', text: 'member' }),
|
h('option', { value: 'member', text: 'member' }),
|
||||||
h('option', { value: 'owner', text: 'owner' }));
|
h('option', { value: 'owner', text: 'owner' }));
|
||||||
const form = h('form', { class: 'inline-form' }, pick, role,
|
const form = h('form', { class: 'inline-form' }, pick, role,
|
||||||
h('button', { class: 'btn', type: 'submit', text: 'Add' }));
|
h('button', { class: 'btn btn-primary', type: 'submit', text: 'Add' }));
|
||||||
form.addEventListener('submit', (e) => {
|
form.addEventListener('submit', (e) => {
|
||||||
e.preventDefault();
|
e.preventDefault();
|
||||||
act(() => api.addTeamMember(Number(pick.value), userID, role.value));
|
act(() => api.addTeamMember(Number(pick.value), userID, role.value));
|
||||||
|
|||||||
@@ -2,8 +2,8 @@
|
|||||||
// incident it belongs to, which is where anything can be done about it.
|
// incident it belongs to, which is where anything can be done about it.
|
||||||
|
|
||||||
import * as api from './api.js';
|
import * as api from './api.js';
|
||||||
import { h, clear, badge, severityBadge, emptyState, spinner } from './ui.js';
|
import { h, clear, badge, severityBadge, originChip, emptyState, spinner } from './ui.js';
|
||||||
import { age, labelSummary } from './format.js';
|
import { age, labelSummary, originOf, ORIGIN_LABEL } from './format.js';
|
||||||
|
|
||||||
const FILTERS = [
|
const FILTERS = [
|
||||||
{ id: 'firing', label: 'Firing', query: { status: 'firing' } },
|
{ id: 'firing', label: 'Firing', query: { status: 'firing' } },
|
||||||
@@ -77,7 +77,8 @@ function row(a) {
|
|||||||
const summary = (a.annotations && a.annotations.summary) || '';
|
const summary = (a.annotations && a.annotations.summary) || '';
|
||||||
const sev = a.labels && a.labels.severity;
|
const sev = a.labels && a.labels.severity;
|
||||||
const labels = labelSummary(Object.fromEntries(
|
const labels = labelSummary(Object.fromEntries(
|
||||||
Object.entries(a.labels || {}).filter(([k]) => k !== 'severity')));
|
Object.entries(a.labels || {}).filter(([k]) => k !== 'severity' && k !== ORIGIN_LABEL)));
|
||||||
|
const origin = originOf(a.labels);
|
||||||
const linked = a.incident_id != null;
|
const linked = a.incident_id != null;
|
||||||
return h(linked ? 'a' : 'div', {
|
return h(linked ? 'a' : 'div', {
|
||||||
class: `row st-${a.status} ${linked ? '' : 'no-link'}`,
|
class: `row st-${a.status} ${linked ? '' : 'no-link'}`,
|
||||||
@@ -86,6 +87,7 @@ function row(a) {
|
|||||||
h('div', { class: 'row-title', text: a.name }),
|
h('div', { class: 'row-title', text: a.name }),
|
||||||
h('div', { class: 'row-age', title: a.starts_at, text: age(a.status === 'firing' ? a.starts_at : a.received_at) }),
|
h('div', { class: 'row-age', title: a.starts_at, text: age(a.status === 'firing' ? a.starts_at : a.received_at) }),
|
||||||
h('div', { class: 'row-meta' },
|
h('div', { class: 'row-meta' },
|
||||||
|
origin && originChip(origin),
|
||||||
badge(a.status === 'firing' ? 'Firing' : 'Resolved', `st-${a.status}`),
|
badge(a.status === 'firing' ? 'Firing' : 'Resolved', `st-${a.status}`),
|
||||||
sev && severityBadge(sev),
|
sev && severityBadge(sev),
|
||||||
summary && h('span', { text: summary }),
|
summary && h('span', { text: summary }),
|
||||||
|
|||||||
@@ -83,6 +83,7 @@ export const setNotifyTarget = (id, ntfyTopic) =>
|
|||||||
|
|
||||||
// incidents
|
// incidents
|
||||||
export const incidents = (query, opts) => call('GET', '/incidents', { query, ...opts });
|
export const incidents = (query, opts) => call('GET', '/incidents', { query, ...opts });
|
||||||
|
export const clusters = (query) => call('GET', '/incidents/clusters', { query });
|
||||||
export const incident = (id) => call('GET', `/incidents/${id}`);
|
export const incident = (id) => call('GET', `/incidents/${id}`);
|
||||||
export const timeline = (id) => call('GET', `/incidents/${id}/timeline`);
|
export const timeline = (id) => call('GET', `/incidents/${id}/timeline`);
|
||||||
|
|
||||||
|
|||||||