Rewrite the README as highlights with screenshots; move the detail into docs/

The README was 1,240 lines of reference material and still described a
SQLite quick start. It is now a short tour (highlights, screenshots of the
web UI, an accurate quick start against Postgres), and each topic has its
own page under docs/ with an index: deployment, configuration, Alertmanager,
incidents, notifications, escalation, dead man's switches, single sign-on,
web UI, API and development. SERVICE-ACCOUNTS.md is rewritten from a
proposal into a reference, and TEAM-LOOKUP.md is gone with the endpoint it
described. The "Upgrading to ..." sections for an unreleased product are
dropped.

Claude-Session: https://claude.ai/code/session_016mBLURvJoMuUEr9cB2RpUN
This commit is contained in:
Niklas Ye
2026-10-09 14:56:13 +02:00
parent 1cb09525e3
commit 44b2eb2cc3
45 changed files with 1355 additions and 2535 deletions
-182
View File
@@ -1,182 +0,0 @@
-- The Postgres baseline: the schema as it stood at the end of the SQLite line,
-- in one file rather than ten.
--
-- The ten SQLite migrations are in git history up to the commit that introduced
-- this one, and they replay against nothing here: their shape was incremental
-- (columns added, then dropped again in 008) and 008's backfill rewrote data
-- that a Postgres install never had. An existing SQLite database is carried over
-- by scripts/sqlite-to-postgres.go, which copies rows into this schema.
--
-- Two conventions inherited deliberately:
--
-- * Timestamps are BIGINT unix seconds, not timestamptz. Everything in Go
-- already speaks epochs, and converting was a second change riding along
-- with the port. Worth revisiting on its own.
--
-- * Ids are GENERATED BY DEFAULT, not ALWAYS, so the migration script can
-- insert rows with their original ids and keep every foreign key intact.
-- setval at the end of the copy puts the sequences past them.
CREATE TABLE users (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
username TEXT NOT NULL UNIQUE,
email TEXT NOT NULL UNIQUE,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
-- Where this user's notifications go. NULL means they get none; incidents
-- assigned to them fall back to the configured fallback topic.
ntfy_topic TEXT,
-- NULL means the user has no password and can only use API keys.
password_hash TEXT
);
CREATE TABLE api_keys (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
key_hash TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
last_used_at BIGINT
);
-- A session is a browser's credential, the cookie counterpart of an API key:
-- only the hash of the token is stored. expires_at slides forward while the
-- session is in use, so an on-call phone stays signed in.
CREATE TABLE sessions (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
token_hash TEXT NOT NULL UNIQUE,
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
created_at BIGINT NOT NULL,
last_seen_at BIGINT NOT NULL,
expires_at BIGINT NOT NULL,
user_agent TEXT
);
CREATE INDEX idx_sessions_user ON sessions(user_id);
-- The machine-owned signal record: what Alertmanager says is true right now.
-- Workflow state lives on incidents, never here, because the webhook upsert owns
-- these rows and would overwrite it.
CREATE TABLE alerts (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
fingerprint TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
status TEXT NOT NULL CHECK (status IN ('firing', 'resolved')),
labels JSONB NOT NULL DEFAULT '{}'::jsonb,
annotations JSONB NOT NULL DEFAULT '{}'::jsonb,
starts_at BIGINT NOT NULL,
ends_at BIGINT,
generator_url TEXT NOT NULL DEFAULT '',
received_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
archived_at BIGINT,
-- Why the alert left the firing state: 'alertmanager' when a resolved
-- webhook set it, 'expiry' when the sweeper inferred it from staleness.
resolution_source TEXT
);
CREATE INDEX alerts_status_idx ON alerts(status);
CREATE INDEX alerts_name_idx ON alerts(name);
CREATE INDEX alerts_received_at_idx ON alerts(received_at DESC);
CREATE INDEX alerts_archived_at_idx ON alerts(archived_at);
CREATE TABLE schedule_entries (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
date TEXT NOT NULL UNIQUE, -- YYYY-MM-DD; one person per day
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
);
CREATE INDEX schedule_entries_date_idx ON schedule_entries(date);
-- The human work item: what people acknowledge, assign, snooze, discuss and
-- resolve. Correlation uses Alertmanager's own groupKey, so incidents follow the
-- group_by routing tree the operator already tuned.
CREATE TABLE incidents (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
group_key TEXT NOT NULL, -- Alertmanager groupKey, opaque
title TEXT NOT NULL, -- rendered from group_labels
group_labels JSONB NOT NULL DEFAULT '{}'::jsonb,
status TEXT NOT NULL CHECK (status IN ('triggered', 'acknowledged', 'resolved')),
severity TEXT, -- highest `severity` label across firing members
triggered_at BIGINT NOT NULL,
acknowledged_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
acknowledged_at BIGINT,
assigned_to BIGINT REFERENCES users(id) ON DELETE SET NULL,
snoozed_until BIGINT,
resolved_at BIGINT,
resolution_source TEXT, -- 'alerts' | 'manual'
archived_at BIGINT
);
-- Load-bearing: at most one OPEN incident per group_key. This is what makes
-- "resolved incident + a new alert occurrence = a new incident" work, and it is
-- the constraint the webhook's find-or-open lookup relies on.
CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents(group_key) WHERE resolved_at IS NULL;
CREATE INDEX incidents_status_idx ON incidents(status);
CREATE INDEX incidents_triggered_at_idx ON incidents(triggered_at DESC);
CREATE INDEX incidents_archived_at_idx ON incidents(archived_at);
-- Membership is historical, not a pointer on alerts: one alert row (one
-- fingerprint) resolves and re-fires over time and belongs to a different
-- incident each occurrence.
CREATE TABLE incident_alerts (
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
alert_id BIGINT NOT NULL REFERENCES alerts(id) ON DELETE CASCADE,
added_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
PRIMARY KEY (incident_id, alert_id)
);
CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts(alert_id);
-- The timeline. Append-only, and the only history this server keeps: alert rows
-- are mutated in place, so without this there is no record that anything
-- happened. Notes are events too, so one query renders the whole story.
CREATE TABLE incident_events (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
-- triggered | alert_added | alert_resolved | acknowledged | unacknowledged
-- | assigned | snoozed | unsnoozed | resolved | note | notified | notify_failed
type TEXT NOT NULL,
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, -- NULL = the server acted
alert_id BIGINT REFERENCES alerts(id) ON DELETE SET NULL,
detail TEXT,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
);
CREATE INDEX incident_events_incident_idx ON incident_events(incident_id, created_at);
-- Delivery is an outbox rather than an inline HTTP call: a POST made while
-- holding the webhook's transaction would hold a connection open across a
-- network round trip. The webhook inserts a row; the notifier goroutine
-- delivers it.
CREATE TABLE notifications (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
-- Nullable: a notification sent to the fallback topic belongs to nobody,
-- because nobody was on call when the incident opened.
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
topic TEXT NOT NULL, -- resolved at enqueue: who was on call then
kind TEXT NOT NULL CHECK (kind IN ('triggered', 'reminder', 'resolved')),
created_at BIGINT NOT NULL,
send_after BIGINT NOT NULL, -- retry backoff watermark
attempts BIGINT NOT NULL DEFAULT 0,
sent_at BIGINT,
last_error TEXT -- kept after the last attempt, for debugging
);
-- The delivery loop's only query: what is due and still unsent.
CREATE INDEX notifications_pending_idx ON notifications(send_after) WHERE sent_at IS NULL;
-- Reminders and resolved notices both look up an incident's newest row.
CREATE INDEX notifications_incident_idx ON notifications(incident_id, id DESC);
-- A notification body is stored on the ntfy server and cached on the device, so
-- a real API key must never appear in one. Each delivery mints its own token
-- instead: one incident, one action, one day.
CREATE TABLE incident_ack_tokens (
token_hash TEXT PRIMARY KEY, -- SHA-256 of the raw token, as with api_keys
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
created_at BIGINT NOT NULL,
expires_at BIGINT NOT NULL
);
CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens(expires_at);
-25
View File
@@ -1,25 +0,0 @@
-- A system administrator role, and the first thing in this server that one user
-- can do and another cannot.
--
-- Until now every authenticated caller could create and delete users, set
-- anybody's password and mint API keys for anybody — auth.go said so in a
-- comment. That was defensible with one operator and a hand-made account; it is
-- not once people sign themselves up (see #7).
--
-- EVERY EXISTING USER BECOMES AN ADMIN. They already hold these powers, so
-- this migration changes nobody's access: it names what is already true, and
-- leaves demotion as a deliberate act somebody performs afterwards. The
-- alternative — promoting only user 1 — would silently strip the others, and
-- could leave an install whose only admin is an account nobody has a password
-- for.
--
-- New users are not admins: the column defaults to false, and the only ways to
-- become one are this backfill, the bootstrap endpoint, or an existing admin
-- granting it.
ALTER TABLE users ADD COLUMN is_admin BOOLEAN NOT NULL DEFAULT false;
UPDATE users SET is_admin = true;
-- The queue's assignment dropdown and the on-call schedule read every user, and
-- the admin screens in #5 will filter on this.
CREATE INDEX users_is_admin_idx ON users(is_admin) WHERE is_admin;
-103
View File
@@ -1,103 +0,0 @@
-- Teams: the unit of tenancy. Everything a person works on now belongs to one.
--
-- Until this migration the install was one shared space — every user saw every
-- alert and every incident, and the Alertmanager webhook was unauthenticated, so
-- anything that could reach the port could open an incident for everybody.
--
-- The shape, in one paragraph: a team owns its incidents, alerts, schedule and
-- integrations. A user belongs to as many teams as they like, with a role in
-- each: an `owner` configures the team, a `member` works its incidents. An
-- integration key is what an alert arrives on, and the key is what says which
-- team the alert belongs to.
--
-- EVERYTHING EXISTING MOVES INTO ONE DEFAULT TEAM, and every existing user
-- becomes an owner of it. That keeps an upgrade a no-op for the people using it:
-- the same queue, the same schedule, the same incidents, with a name on them.
CREATE TABLE teams (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
name TEXT NOT NULL UNIQUE,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
);
-- role is free text with a CHECK rather than an enum, so adding a third role
-- later is a migration and not a type rewrite.
CREATE TABLE team_members (
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
joined_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
PRIMARY KEY (team_id, user_id)
);
CREATE INDEX team_members_user_idx ON team_members(user_id);
-- How alerts get in, and the only thing that says which team they belong to.
-- The key is stored as a SHA-256 hash, like api_keys and the ack tokens: a
-- leaked database gives nobody the ability to post alerts.
CREATE TABLE integrations (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
kind TEXT NOT NULL CHECK (kind IN ('alertmanager')),
name TEXT NOT NULL,
key_hash TEXT NOT NULL UNIQUE,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
last_used_at BIGINT
);
CREATE INDEX integrations_team_idx ON integrations(team_id);
-- ---------------------------------------------------------------------------
-- The default team, and everything that already exists moving into it.
--
-- Created unconditionally, even on an empty install, so there is always a team
-- for the bootstrap user to land in and for the first integration to hang off.
-- ---------------------------------------------------------------------------
INSERT INTO teams (name) VALUES ('Default');
INSERT INTO team_members (team_id, user_id, role)
SELECT (SELECT id FROM teams WHERE name = 'Default'), id, 'owner' FROM users;
-- ---------------------------------------------------------------------------
-- team_id on everything a team owns.
--
-- Added nullable, backfilled, then made NOT NULL: adding a NOT NULL column with
-- no default to a table with rows is rejected, and a DEFAULT pointing at the
-- default team would quietly keep working after the default team is gone.
-- ---------------------------------------------------------------------------
ALTER TABLE alerts ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
ALTER TABLE incidents ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
ALTER TABLE schedule_entries ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
UPDATE alerts SET team_id = (SELECT id FROM teams WHERE name = 'Default');
UPDATE incidents SET team_id = (SELECT id FROM teams WHERE name = 'Default');
UPDATE schedule_entries SET team_id = (SELECT id FROM teams WHERE name = 'Default');
ALTER TABLE alerts ALTER COLUMN team_id SET NOT NULL;
ALTER TABLE incidents ALTER COLUMN team_id SET NOT NULL;
ALTER TABLE schedule_entries ALTER COLUMN team_id SET NOT NULL;
-- ---------------------------------------------------------------------------
-- The uniqueness rules were all written for one tenant, and every one of them
-- is wrong now: two teams monitoring two clusters legitimately see the same
-- fingerprint, the same groupKey, and want somebody on call on the same day.
-- ---------------------------------------------------------------------------
ALTER TABLE alerts DROP CONSTRAINT alerts_fingerprint_key;
CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts(team_id, fingerprint);
DROP INDEX incidents_open_group_key_idx;
-- Still load-bearing, now per team: at most one OPEN incident per group_key
-- within a team. This is what makes "resolved incident + a new alert occurrence
-- = a new incident" work, and what the webhook's find-or-open lookup relies on.
CREATE UNIQUE INDEX incidents_open_group_key_idx
ON incidents(team_id, group_key) WHERE resolved_at IS NULL;
ALTER TABLE schedule_entries DROP CONSTRAINT schedule_entries_date_key;
CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries(team_id, date);
-- The list views all filter by team first.
CREATE INDEX alerts_team_received_idx ON alerts(team_id, received_at DESC);
CREATE INDEX incidents_team_triggered_idx ON incidents(team_id, triggered_at DESC);
@@ -1,39 +0,0 @@
-- Dead man's switches become a team's own configuration.
--
-- They were three environment variables — TERDUT_DEADMAN_MATCHERS, _TIMEOUT and
-- _SEVERITY — which made them one setting for the whole install. That was the
-- last piece of the alerting path a team could not control: a team could take
-- its own alerts on its own key and still not say which of them were
-- heartbeats, or how long a silence had to last before somebody was paged.
--
-- One row per team rather than one row per switch. The unit of monitoring is
-- still the fingerprint, as it always was — two clusters sending the same
-- heartbeat alertname are two independent switches — and the matcher string
-- keeps the format the environment variable used, so a value can be moved from
-- one to the other unchanged.
--
-- No rows are seeded here: a migration cannot read the environment. The server
-- inserts a row per team at startup from its own configuration, and the same
-- values therefore carry forward into the first team's row without anybody
-- retyping them. See seedDeadmanConfigs.
CREATE TABLE deadman_configs (
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
-- ";" separates matchers, "," the label conditions within one, "=" is exact
-- equality: `alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat`.
-- Every matcher must name an alertname. Empty watches nothing.
matchers TEXT NOT NULL DEFAULT '',
-- Seconds rather than a Go duration string: the column is compared and
-- arithmetic is done on it, and a value that has to be parsed before it can
-- be believed is a value that can be stored unparseable. Zero disables the
-- team's switches entirely.
timeout_seconds BIGINT NOT NULL DEFAULT 0,
-- The severity these incidents open at. They have no member alerts to
-- derive one from, and a heartbeat's own severity label is meaningless —
-- Watchdog ships as "none".
severity TEXT NOT NULL DEFAULT 'critical',
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
);
-35
View File
@@ -1,35 +0,0 @@
-- Settings that an administrator can change without a redeploy, and the flag
-- that takes an account out of use without deleting it.
--
-- Three of the server's tunables were environment variables, which meant
-- changing how long an incident waits before it is paged again required editing
-- a chart, merging it, and waiting for a reconcile. They are behaviour, not
-- infrastructure, and the difference is who needs to change them and how often.
--
-- What stays in the environment: the ntfy URL and token, the database DSN, the
-- listen address and the public URL. Those are where the server is plugged in
-- rather than how it behaves, they are needed before the database is open, and
-- two of them are credentials.
--
-- Key/value rather than a column per setting. A settings table with one row and
-- a column per knob needs a migration for every new knob, and #6 and #7 will
-- both add some. The cost is that values are text and the accessor has to say
-- what type it wanted; settings.go does that in one place.
--
-- No rows are seeded here: a migration cannot read the environment. The server
-- inserts each key from its own configuration at startup, once, so an install
-- that upgrades keeps exactly the behaviour it had. See SeedSettings.
CREATE TABLE settings (
key TEXT PRIMARY KEY,
value TEXT NOT NULL,
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
);
-- Disabling an account rather than deleting it: the person has left, or the
-- credential is suspect, and their incidents, acknowledgements and timeline
-- entries must stay exactly where they are. Deleting a user nulls their
-- acknowledged_by and assigned_to, which quietly rewrites history.
--
-- A disabled user cannot sign in and their API keys stop working, but they are
-- still a name the timeline can show and still a member of their teams.
ALTER TABLE users ADD COLUMN disabled_at BIGINT;
-95
View File
@@ -1,95 +0,0 @@
-- Escalation: page somebody else when the first person does not answer.
--
-- This is the gap the whole multi-tenancy line of work was opened to close.
-- Until now an unacknowledged incident re-paged the same topic every
-- notify_repeat forever, which is a louder version of the same silence: if the
-- person on call is asleep, has no signal, or has left, nothing else happens.
--
-- Shape: one policy per team, an ordered list of levels, each level with a
-- timeout and a set of targets. When a level's timeout passes and the incident
-- is still triggered, the next level is paged. When the last level passes, the
-- chain repeats repeat_count times, and then the team's fallback topic is paged
-- once as the end of the line.
--
-- A team WITHOUT a policy keeps exactly today's behaviour: page the assignee,
-- then remind on the same topic. Escalation is opt-in per team, and the two
-- never both run for one incident -- see enqueueReminders.
CREATE TABLE escalation_policies (
-- One per team for now, hence the team as the key rather than an id with a
-- unique index: routing different alerts to different chains needs the
-- alert to carry something to route ON, which is a separate question.
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
-- How many extra times to run the whole chain after it has been walked
-- once. 0 means walk it once and stop at the fallback.
repeat_count BIGINT NOT NULL DEFAULT 0 CHECK (repeat_count >= 0 AND repeat_count <= 10),
-- Where the last page goes when every level has been tried. Per team now:
-- TERDUT_NTFY_FALLBACK_TOPIC was one topic for the whole install, which in
-- a multi-team server pages the wrong people. Empty means the chain simply
-- ends.
fallback_topic TEXT NOT NULL DEFAULT '',
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
);
CREATE TABLE escalation_levels (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
team_id BIGINT NOT NULL REFERENCES escalation_policies(team_id) ON DELETE CASCADE,
-- 1-based, dense. The API rewrites the whole ladder on every edit rather
-- than patching one rung, so there is no way to leave a gap.
position BIGINT NOT NULL,
-- How long this level has to produce an acknowledgement before the next one
-- is paged. Seconds, like every other duration in this schema.
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
UNIQUE (team_id, position)
);
-- Who a level pages. Either a named person, or whoever the team's rota says is
-- on call today -- which is the target that keeps working when the rota
-- changes and nobody remembers to edit the policy.
CREATE TABLE escalation_targets (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
level_id BIGINT NOT NULL REFERENCES escalation_levels(id) ON DELETE CASCADE,
kind TEXT NOT NULL CHECK (kind IN ('user', 'oncall')),
-- Set for kind='user', NULL for kind='oncall'.
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
CHECK ((kind = 'user' AND user_id IS NOT NULL) OR (kind = 'oncall' AND user_id IS NULL))
);
CREATE INDEX escalation_targets_level_idx ON escalation_targets(level_id);
-- ---------------------------------------------------------------------------
-- Where an incident is in its chain.
--
-- On the incident rather than in a side table: it is read on every notifier
-- tick alongside the incident's status, and one row per incident is exactly
-- what the state is.
-- ---------------------------------------------------------------------------
-- 0 means no level has been paged yet, which is the state of every incident
-- that existed before escalation and of every incident in a team with no
-- policy. 1 is the first level.
ALTER TABLE incidents ADD COLUMN escalation_level BIGINT NOT NULL DEFAULT 0;
-- When the current level was entered, and therefore what its timeout is
-- measured from. NULL while escalation_level is 0.
ALTER TABLE incidents ADD COLUMN escalation_level_at BIGINT;
-- How many times the chain has been walked in full. Compared against the
-- policy's repeat_count.
ALTER TABLE incidents ADD COLUMN escalation_round BIGINT NOT NULL DEFAULT 0;
-- The notifier's escalation query: incidents still waiting, oldest level first.
CREATE INDEX incidents_escalation_idx
ON incidents(escalation_level_at)
WHERE resolved_at IS NULL AND status = 'triggered';
-- 'escalated' joins the outbox kinds: a page that went out because nobody
-- answered the last one, which is worth telling apart from the first page and
-- from a reminder when reading the timeline or debugging a delivery.
ALTER TABLE notifications DROP CONSTRAINT notifications_kind_check;
ALTER TABLE notifications ADD CONSTRAINT notifications_kind_check
CHECK (kind IN ('triggered', 'reminder', 'resolved', 'escalated'));
@@ -1,49 +0,0 @@
-- Self-service sign-up, and the invite links that make it useful.
--
-- Until now the only way to get an account was for somebody who already had one
-- to create it, and the login page told people to "ask an admin". That is a
-- workable arrangement for one operator and an impossible one for a team.
--
-- An invite is a link, not an email: this server has no SMTP and adding it to
-- send one message would be a new subsystem to run, secure and monitor. The
-- person inviting sends the link however they already talk to the person they
-- are inviting.
CREATE TABLE invites (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
-- SHA-256 of the raw token, like api_keys, the integration keys and the
-- acknowledgement tokens. A leaked database hands nobody an account.
token_hash TEXT NOT NULL UNIQUE,
-- Which team the invitee lands in, and as what. An invite always names a
-- team: an account in no team sees an empty queue and can be paged by
-- nobody, which is not a state to invite somebody into.
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
-- Invites expire. A link that works forever is a credential nobody
-- remembers issuing, sitting in a chat log.
expires_at BIGINT NOT NULL,
-- Single-use by default: max_uses 1. A team onboarding six people at once
-- can raise it rather than minting six links.
max_uses BIGINT NOT NULL DEFAULT 1 CHECK (max_uses > 0 AND max_uses <= 100),
uses BIGINT NOT NULL DEFAULT 0,
-- Revoked by hand, separately from expiry, so "this link is no longer
-- wanted" and "this link timed out" stay distinguishable in the listing.
revoked_at BIGINT
);
CREATE INDEX invites_team_idx ON invites(team_id);
-- Who redeemed which invite. Kept after the invite is gone — the answer to "how
-- did this account get here" should outlive the link that made it.
ALTER TABLE users ADD COLUMN invited_via BIGINT REFERENCES invites(id) ON DELETE SET NULL;
-- Where a person is in the first-run checklist, so it can be resumed and
-- dismissed rather than nagging forever. One row per user, created on demand.
ALTER TABLE users ADD COLUMN onboarding_dismissed_at BIGINT;
@@ -1,23 +0,0 @@
-- Similar incidents: a signature per incident, so "has this happened before"
-- is an indexed equality instead of a search.
--
-- The signature is the alert name plus the group labels that identify WHAT is
-- broken, minus the ones that only say WHERE it happened to run this time
-- (instance, pod, ...). Two incidents with the same signature in the same team
-- are the same problem for a responder's purposes.
--
-- Computed in Go for new incidents (incidentSignature in incident_store.go).
-- The backfill below MUST produce the same string; keep the volatile list in
-- both places in step.
ALTER TABLE incidents ADD COLUMN signature TEXT NOT NULL DEFAULT '';
UPDATE incidents SET signature =
COALESCE(NULLIF(group_labels->>'alertname', ''), title) || '|' ||
COALESCE((
SELECT string_agg(e.k || '=' || e.v, ',' ORDER BY e.k)
FROM jsonb_each_text(incidents.group_labels) AS e(k, v)
WHERE e.k <> 'alertname'
AND e.k NOT IN ('instance', 'pod', 'pod_name', 'pod_ip', 'container', 'container_name', 'endpoint')
), '');
CREATE INDEX incidents_signature_idx ON incidents(team_id, signature, triggered_at DESC);
@@ -1,54 +0,0 @@
-- Dead man's switches become rows of their own.
--
-- 004 kept a team's switches in one string with one timeout and one severity,
-- which was enough to configure them and not enough to show them: there was no
-- thing to list, nothing to hang a status on, and every switch in a team had to
-- share a deadline. A row per switch gives each its own name, matcher, timeout
-- and severity, and gives the Team → Switches page something to be a list of.
--
-- The matcher keeps the syntax the string used, one matcher per row:
-- `alertname=Watchdog,cluster=prod`. The unit of monitoring is still the
-- fingerprint, so a matcher that many clusters satisfy is still one switch row
-- watching several independent heartbeats.
CREATE TABLE deadman_switches (
id BIGSERIAL PRIMARY KEY,
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
-- What the owner calls it. Defaults to the matcher when they do not say.
name TEXT NOT NULL,
-- "," separates the label conditions, "=" is exact equality, and alertname is
-- mandatory: it is what keeps the sweeper's candidate query on an index.
matcher TEXT NOT NULL,
-- Seconds of silence before the switch is declared dead. Never zero: a switch
-- that cannot fire is deleted, not disabled.
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
-- The severity its incidents open at. See 004 for why they carry their own.
severity TEXT NOT NULL DEFAULT 'critical',
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
);
CREATE INDEX deadman_switches_team_idx ON deadman_switches (team_id);
-- Carry every team's configuration over, one row per matcher. A team whose
-- timeout was zero had switches turned off, which is now "no rows".
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
SELECT c.team_id, btrim(m), btrim(m), c.timeout_seconds, c.severity
FROM deadman_configs c,
LATERAL regexp_split_to_table(c.matchers, ';') AS m
WHERE c.timeout_seconds > 0
AND btrim(m) <> ''
ORDER BY c.team_id;
-- The server seeds environment defaults into teams once, and remembers that it
-- did. An install that had a row per team was already seeded; without this
-- marker the first start after upgrading would seed teams that had switched
-- theirs off.
INSERT INTO settings (key, value)
SELECT 'deadman_seeded', '1'
WHERE EXISTS (SELECT 1 FROM deadman_configs);
DROP TABLE deadman_configs;
@@ -1,21 +0,0 @@
-- Which alert source an alert last arrived on.
--
-- Team -> Sources shows when each source last posted, which integrations
-- already knew (last_used_at, stamped on every webhook). What it could not say
-- was what a source delivered: an alert never recorded the key it came in on, so
-- "prod alertmanager" and "staging alertmanager" were indistinguishable once
-- inside. This column is that link, and lets the page show each source's last
-- alert and how many alerts it has kept fresh over the past day.
--
-- Last sender wins: every accepted payload restamps it, the way it advances
-- received_at. Two sources posting the same fingerprint into one team is
-- already one alert, and it is attributed to whichever spoke last.
--
-- Nullable, and not backfilled. Alerts that arrived before this migration have
-- no source, and NULL says so honestly rather than guessing. It heals by itself:
-- Alertmanager re-sends every alert each repeat_interval, and each re-send is an
-- accepted payload. Deleting a source keeps its alerts, unattributed.
ALTER TABLE alerts ADD COLUMN integration_id BIGINT REFERENCES integrations(id) ON DELETE SET NULL;
CREATE INDEX alerts_integration_idx ON alerts (integration_id, received_at)
WHERE integration_id IS NOT NULL;
-60
View File
@@ -1,60 +0,0 @@
-- Single sign-on through an OpenID Connect provider (Authentik, and anything
-- else that speaks OIDC).
--
-- Four things change, and none of them touches a password user: every new column
-- has a default that says "this is how it has always worked".
--
-- 1. user_identities says which provider account a user is. It is keyed on
-- (issuer, subject), never on email or username: those are mutable at the
-- provider, and a recycled address must not inherit somebody's account. A
-- user can have several identities (a second provider later), and none at all
-- (a local, password-only user), which is why this is a table and not two
-- columns on users.
--
-- 2. team_members.source and users.admin_source record who granted a role. 'oidc'
-- rows are owned by the group sync: it adds them when a group grants access
-- and removes them when it stops, and nothing else may edit them. 'manual' rows
-- are everything that existed before this migration, and are never touched by
-- the sync. Without the marker the sync could not tell a membership it created
-- from one an owner added by hand, and would have to either leave stale access
-- behind or delete people it had no business deleting.
--
-- 3. sessions.max_expires_at is a hard ceiling on a session's life. Ordinary
-- sessions slide for as long as they are used; a session made by an SSO login
-- must not, because the login is the only moment the groups are re-read.
-- Capping the session is what makes "removed from the group in the provider"
-- take effect within a bounded time. NULL means no ceiling.
--
-- 4. oidc_logins holds a login that has been started and not yet finished: the
-- state, nonce and PKCE verifier the callback must see again. A row rather
-- than a signed cookie, so it survives a restart and needs no signing key.
-- Only the hash of the state is stored, like every other token here; the
-- nonce and verifier are useless without the state that names the row.
CREATE TABLE user_identities (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
issuer TEXT NOT NULL,
subject TEXT NOT NULL,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
last_login_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
UNIQUE (issuer, subject)
);
CREATE INDEX user_identities_user_idx ON user_identities (user_id);
ALTER TABLE team_members
ADD COLUMN source TEXT NOT NULL DEFAULT 'manual' CHECK (source IN ('manual', 'oidc'));
ALTER TABLE users
ADD COLUMN admin_source TEXT NOT NULL DEFAULT 'manual' CHECK (admin_source IN ('manual', 'oidc'));
ALTER TABLE sessions ADD COLUMN max_expires_at BIGINT;
CREATE TABLE oidc_logins (
state_hash TEXT PRIMARY KEY,
nonce TEXT NOT NULL,
pkce_verifier TEXT NOT NULL,
expires_at BIGINT NOT NULL
);
CREATE INDEX oidc_logins_expires_idx ON oidc_logins (expires_at);
@@ -1,40 +0,0 @@
-- Signing in from a terminal, for clients that cannot open a browser on the
-- machine they run on (the TUI over SSH is the reason).
--
-- The flow is the OAuth device authorization grant, run by terdut itself rather
-- than the identity provider, so the terminal never talks to the provider and
-- the server issues its ordinary session at the end:
--
-- 1. The terminal asks for a login and gets two secrets: a device code it
-- keeps and polls with, and a short user code it shows the person.
-- 2. The person opens the verification URL on any device, signs in by whatever
-- means the server offers, sees the user code, and approves it.
-- 3. The terminal's next poll finds the row approved and is given a session.
--
-- Only the hash of the device code is stored, like every other token here: the
-- device code is what earns a session, so a database read must not yield one.
-- The user code is shown on screens and typed by people, so it is stored as is;
-- on its own it can only be approved, never redeemed.
--
-- user_id is the person who approved. It is empty until then, and the session
-- is minted at redemption, not at approval: an approval nobody collects must not
-- leave a live session lying about.
--
-- last_polled_at lets the server refuse a client that polls faster than the
-- interval it was told.
CREATE TABLE device_logins (
device_hash TEXT PRIMARY KEY,
user_code TEXT NOT NULL UNIQUE,
status TEXT NOT NULL DEFAULT 'pending' CHECK (status IN ('pending', 'approved', 'denied')),
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
expires_at BIGINT NOT NULL,
last_polled_at BIGINT NOT NULL DEFAULT 0
);
CREATE INDEX device_logins_expires_idx ON device_logins (expires_at);
-- Where to send the browser once a single sign-on login completes. A person who
-- opens /device?code=... without a session has to sign in first and then come
-- back to it, and the same is true of any other deep link. Validated when it is
-- stored: only a path on this server is ever kept.
ALTER TABLE oidc_logins ADD COLUMN next TEXT NOT NULL DEFAULT '/';
@@ -1,26 +0,0 @@
-- Per-team OIDC group configuration, replacing the global
-- TERDUT_OIDC_GROUP_MAPPINGS env var.
--
-- Group -> team -> role used to be one global list an operator set for the
-- whole install, matched against a team by name, and the sync would create
-- the team if no team by that name existed yet. That put the decision of
-- which group controls a team in the server's environment rather than the
-- team's own hands, meant changing it needed an env var edit and a restart,
-- and let a typo in a team name silently create a stray team.
--
-- Each team now names, itself, which group grants membership and which
-- grants ownership. Nullable: most teams need neither. No uniqueness
-- constraint on either column — two teams may legitimately watch the same
-- provider group (a broad team and a narrower one both keyed off overlapping
-- groups is a choice for their owners to make, not one the schema should
-- refuse).
--
-- BREAKING CHANGE, deliberately not auto-migrated: TERDUT_OIDC_GROUP_MAPPINGS
-- stops being read as of this version, and the sync no longer creates a team
-- by name. Every team's group binding must be set again through
-- PUT /api/teams/{teamID}/oidc-groups. Until an owner does that, an
-- OIDC-sourced membership in that team is dropped at that user's next SSO
-- sign-in, the same way any other loss of group access is handled. See the
-- README's OIDC section.
ALTER TABLE teams ADD COLUMN oidc_member_group TEXT;
ALTER TABLE teams ADD COLUMN oidc_owner_group TEXT;
@@ -1,43 +0,0 @@
-- Service accounts: a scoped, non-human credential for automation (e.g.
-- terdut-operator) that needs to manage teams, escalation policies, dead
-- man's switches, integrations and OIDC group bindings without impersonating
-- a human user. See SERVICE-ACCOUNTS.md for the design this implements.
--
-- Deliberately not a users row: no password_hash, no is_admin, no
-- user_identities linkage, so a service account can never be pulled into
-- OIDC group sync or password login, and is never mistaken for a human in an
-- audit trail.
--
-- scope is 'instance' (acts with the same reach system administration has
-- over teams: create one, list them, mint a 'team'-scoped account against
-- any of them) or 'team' (acts as that one team's owner, and nothing else).
-- The CHECK ties team_id's presence to scope directly, rather than leaving it
-- to application code to keep the two consistent.
CREATE TABLE service_accounts (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
name TEXT NOT NULL UNIQUE,
scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
CONSTRAINT service_accounts_scope_team_id_chk CHECK (
(scope = 'team' AND team_id IS NOT NULL) OR
(scope = 'instance' AND team_id IS NULL)
)
);
CREATE INDEX service_accounts_team_id_idx ON service_accounts(team_id);
-- One account, many keys: rotation is minting a new one and revoking the
-- old, the same shape api_keys already has, so an account's identity and
-- audit history survive a rotation instead of being recreated by it.
CREATE TABLE service_account_keys (
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
key_hash TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
last_used_at BIGINT
);
CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys(service_account_id);
@@ -1,39 +0,0 @@
-- Service-account actors on incident mutations (terdut-server#25). A
-- team-scoped service account acknowledging/resolving/snoozing/noting an
-- incident is not a users row, so it cannot be written into
-- acknowledged_by/incident_events.user_id — doing so either violates the
-- users(id) FK (new rows) or, for incident_events.user_id, silently matches
-- zero rows on delete. These columns are the service-account-shaped parallel
-- to the existing human ones: nullable, mutually exclusive with their human
-- counterpart, ON DELETE SET NULL so a deleted service account doesn't take
-- the incident history with it.
ALTER TABLE incidents
ADD COLUMN acknowledged_by_service_account_id BIGINT
REFERENCES service_accounts(id) ON DELETE SET NULL;
ALTER TABLE incident_events
ADD COLUMN service_account_id BIGINT
REFERENCES service_accounts(id) ON DELETE SET NULL;
-- At most one actor kind per row: both NULL ("the server acted") is valid,
-- exactly one set is valid, both set is a bug this constraint refuses to
-- store rather than silently accepting.
ALTER TABLE incidents
ADD CONSTRAINT incidents_ack_actor_xor_chk CHECK (
acknowledged_by IS NULL OR acknowledged_by_service_account_id IS NULL
);
ALTER TABLE incident_events
ADD CONSTRAINT incident_events_actor_xor_chk CHECK (
user_id IS NULL OR service_account_id IS NULL
);
CREATE INDEX incidents_acknowledged_by_service_account_id_idx
ON incidents(acknowledged_by_service_account_id);
CREATE INDEX incident_events_service_account_id_idx
ON incident_events(service_account_id);
-- assigned_to_service_account_id is deliberately not added here: it would sit
-- unpopulated until handleIncidentAssign itself tracks an actor, which is a
-- separate, pre-existing gap (it records the assignee today, never the
-- actor, for humans either) tracked in its own follow-up issue.
@@ -1,16 +0,0 @@
-- Backs the rate limiters (failed logins, sign-ups, OIDC/device start) with
-- Postgres instead of an in-memory map, now that the server runs more than
-- one replica in production (v0.37.0): a counter that only ever sees its own
-- pod's traffic quietly let every one of these limits through multiplied by
-- the replica count.
--
-- window_start is the start of the current fixed window for key, in the same
-- "unix seconds" shape every other timestamp in this schema uses. The window
-- resets rather than slides, matching the in-memory limiter it replaces:
-- once a key's window is older than the limiter's window length, the next
-- failure starts a fresh one instead of extending the stale one.
CREATE TABLE rate_limit_counters (
key TEXT PRIMARY KEY,
window_start BIGINT NOT NULL,
count INT NOT NULL
);
@@ -1,7 +0,0 @@
-- Optional expiry on a user's own API keys. NULL (the existing default for
-- every row already in this table) means "never expires" -- the same
-- behavior these keys have always had, so no existing integration breaks.
-- Service account keys are deliberately NOT touched: they are a different
-- table, managed by automation, and already distinguished by their own
-- "tdsa_" prefix.
ALTER TABLE api_keys ADD COLUMN expires_at BIGINT;
@@ -1,21 +0,0 @@
-- Who performed an assignment (terdut-server#35). On an 'assigned' event
-- incident_events.user_id is the assignee, so the actor needs columns of its
-- own. Only populated for 'assigned' events; every other event type keeps
-- using user_id/service_account_id for the actor. Older 'assigned' rows stay
-- NULL (the actor was never recorded). Same shape as migration 015: nullable,
-- mutually exclusive, ON DELETE SET NULL.
--
-- assigned_to_service_account_id is still deliberately not added: making
-- service accounts assignable is a separate change (request body, assignee
-- picker, notifier, filters).
ALTER TABLE incident_events
ADD COLUMN actor_user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
ADD COLUMN actor_service_account_id BIGINT REFERENCES service_accounts(id) ON DELETE SET NULL;
ALTER TABLE incident_events
ADD CONSTRAINT incident_events_assign_actor_xor_chk CHECK (
actor_user_id IS NULL OR actor_service_account_id IS NULL
);
CREATE INDEX incident_events_actor_user_id_idx ON incident_events(actor_user_id);
CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events(actor_service_account_id);