Rewrite the README as highlights with screenshots; move the detail into docs/
The README was 1,240 lines of reference material and still described a SQLite quick start. It is now a short tour (highlights, screenshots of the web UI, an accurate quick start against Postgres), and each topic has its own page under docs/ with an index: deployment, configuration, Alertmanager, incidents, notifications, escalation, dead man's switches, single sign-on, web UI, API and development. SERVICE-ACCOUNTS.md is rewritten from a proposal into a reference, and TEAM-LOOKUP.md is gone with the endpoint it described. The "Upgrading to ..." sections for an unreleased product are dropped. Claude-Session: https://claude.ai/code/session_016mBLURvJoMuUEr9cB2RpUN
This commit is contained in:
@@ -1,182 +0,0 @@
|
||||
-- The Postgres baseline: the schema as it stood at the end of the SQLite line,
|
||||
-- in one file rather than ten.
|
||||
--
|
||||
-- The ten SQLite migrations are in git history up to the commit that introduced
|
||||
-- this one, and they replay against nothing here: their shape was incremental
|
||||
-- (columns added, then dropped again in 008) and 008's backfill rewrote data
|
||||
-- that a Postgres install never had. An existing SQLite database is carried over
|
||||
-- by scripts/sqlite-to-postgres.go, which copies rows into this schema.
|
||||
--
|
||||
-- Two conventions inherited deliberately:
|
||||
--
|
||||
-- * Timestamps are BIGINT unix seconds, not timestamptz. Everything in Go
|
||||
-- already speaks epochs, and converting was a second change riding along
|
||||
-- with the port. Worth revisiting on its own.
|
||||
--
|
||||
-- * Ids are GENERATED BY DEFAULT, not ALWAYS, so the migration script can
|
||||
-- insert rows with their original ids and keep every foreign key intact.
|
||||
-- setval at the end of the copy puts the sequences past them.
|
||||
|
||||
CREATE TABLE users (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
username TEXT NOT NULL UNIQUE,
|
||||
email TEXT NOT NULL UNIQUE,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
-- Where this user's notifications go. NULL means they get none; incidents
|
||||
-- assigned to them fall back to the configured fallback topic.
|
||||
ntfy_topic TEXT,
|
||||
-- NULL means the user has no password and can only use API keys.
|
||||
password_hash TEXT
|
||||
);
|
||||
|
||||
CREATE TABLE api_keys (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
key_hash TEXT NOT NULL UNIQUE,
|
||||
name TEXT NOT NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
last_used_at BIGINT
|
||||
);
|
||||
|
||||
-- A session is a browser's credential, the cookie counterpart of an API key:
|
||||
-- only the hash of the token is stored. expires_at slides forward while the
|
||||
-- session is in use, so an on-call phone stays signed in.
|
||||
CREATE TABLE sessions (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
token_hash TEXT NOT NULL UNIQUE,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
created_at BIGINT NOT NULL,
|
||||
last_seen_at BIGINT NOT NULL,
|
||||
expires_at BIGINT NOT NULL,
|
||||
user_agent TEXT
|
||||
);
|
||||
|
||||
CREATE INDEX idx_sessions_user ON sessions(user_id);
|
||||
|
||||
-- The machine-owned signal record: what Alertmanager says is true right now.
|
||||
-- Workflow state lives on incidents, never here, because the webhook upsert owns
|
||||
-- these rows and would overwrite it.
|
||||
CREATE TABLE alerts (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
fingerprint TEXT NOT NULL UNIQUE,
|
||||
name TEXT NOT NULL,
|
||||
status TEXT NOT NULL CHECK (status IN ('firing', 'resolved')),
|
||||
labels JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
annotations JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
starts_at BIGINT NOT NULL,
|
||||
ends_at BIGINT,
|
||||
generator_url TEXT NOT NULL DEFAULT '',
|
||||
received_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
archived_at BIGINT,
|
||||
-- Why the alert left the firing state: 'alertmanager' when a resolved
|
||||
-- webhook set it, 'expiry' when the sweeper inferred it from staleness.
|
||||
resolution_source TEXT
|
||||
);
|
||||
|
||||
CREATE INDEX alerts_status_idx ON alerts(status);
|
||||
CREATE INDEX alerts_name_idx ON alerts(name);
|
||||
CREATE INDEX alerts_received_at_idx ON alerts(received_at DESC);
|
||||
CREATE INDEX alerts_archived_at_idx ON alerts(archived_at);
|
||||
|
||||
CREATE TABLE schedule_entries (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
date TEXT NOT NULL UNIQUE, -- YYYY-MM-DD; one person per day
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
CREATE INDEX schedule_entries_date_idx ON schedule_entries(date);
|
||||
|
||||
-- The human work item: what people acknowledge, assign, snooze, discuss and
|
||||
-- resolve. Correlation uses Alertmanager's own groupKey, so incidents follow the
|
||||
-- group_by routing tree the operator already tuned.
|
||||
CREATE TABLE incidents (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
group_key TEXT NOT NULL, -- Alertmanager groupKey, opaque
|
||||
title TEXT NOT NULL, -- rendered from group_labels
|
||||
group_labels JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
status TEXT NOT NULL CHECK (status IN ('triggered', 'acknowledged', 'resolved')),
|
||||
severity TEXT, -- highest `severity` label across firing members
|
||||
triggered_at BIGINT NOT NULL,
|
||||
acknowledged_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
acknowledged_at BIGINT,
|
||||
assigned_to BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
snoozed_until BIGINT,
|
||||
resolved_at BIGINT,
|
||||
resolution_source TEXT, -- 'alerts' | 'manual'
|
||||
archived_at BIGINT
|
||||
);
|
||||
|
||||
-- Load-bearing: at most one OPEN incident per group_key. This is what makes
|
||||
-- "resolved incident + a new alert occurrence = a new incident" work, and it is
|
||||
-- the constraint the webhook's find-or-open lookup relies on.
|
||||
CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents(group_key) WHERE resolved_at IS NULL;
|
||||
CREATE INDEX incidents_status_idx ON incidents(status);
|
||||
CREATE INDEX incidents_triggered_at_idx ON incidents(triggered_at DESC);
|
||||
CREATE INDEX incidents_archived_at_idx ON incidents(archived_at);
|
||||
|
||||
-- Membership is historical, not a pointer on alerts: one alert row (one
|
||||
-- fingerprint) resolves and re-fires over time and belongs to a different
|
||||
-- incident each occurrence.
|
||||
CREATE TABLE incident_alerts (
|
||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
||||
alert_id BIGINT NOT NULL REFERENCES alerts(id) ON DELETE CASCADE,
|
||||
added_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
PRIMARY KEY (incident_id, alert_id)
|
||||
);
|
||||
|
||||
CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts(alert_id);
|
||||
|
||||
-- The timeline. Append-only, and the only history this server keeps: alert rows
|
||||
-- are mutated in place, so without this there is no record that anything
|
||||
-- happened. Notes are events too, so one query renders the whole story.
|
||||
CREATE TABLE incident_events (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
||||
-- triggered | alert_added | alert_resolved | acknowledged | unacknowledged
|
||||
-- | assigned | snoozed | unsnoozed | resolved | note | notified | notify_failed
|
||||
type TEXT NOT NULL,
|
||||
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, -- NULL = the server acted
|
||||
alert_id BIGINT REFERENCES alerts(id) ON DELETE SET NULL,
|
||||
detail TEXT,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
CREATE INDEX incident_events_incident_idx ON incident_events(incident_id, created_at);
|
||||
|
||||
-- Delivery is an outbox rather than an inline HTTP call: a POST made while
|
||||
-- holding the webhook's transaction would hold a connection open across a
|
||||
-- network round trip. The webhook inserts a row; the notifier goroutine
|
||||
-- delivers it.
|
||||
CREATE TABLE notifications (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
||||
-- Nullable: a notification sent to the fallback topic belongs to nobody,
|
||||
-- because nobody was on call when the incident opened.
|
||||
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
topic TEXT NOT NULL, -- resolved at enqueue: who was on call then
|
||||
kind TEXT NOT NULL CHECK (kind IN ('triggered', 'reminder', 'resolved')),
|
||||
created_at BIGINT NOT NULL,
|
||||
send_after BIGINT NOT NULL, -- retry backoff watermark
|
||||
attempts BIGINT NOT NULL DEFAULT 0,
|
||||
sent_at BIGINT,
|
||||
last_error TEXT -- kept after the last attempt, for debugging
|
||||
);
|
||||
|
||||
-- The delivery loop's only query: what is due and still unsent.
|
||||
CREATE INDEX notifications_pending_idx ON notifications(send_after) WHERE sent_at IS NULL;
|
||||
-- Reminders and resolved notices both look up an incident's newest row.
|
||||
CREATE INDEX notifications_incident_idx ON notifications(incident_id, id DESC);
|
||||
|
||||
-- A notification body is stored on the ntfy server and cached on the device, so
|
||||
-- a real API key must never appear in one. Each delivery mints its own token
|
||||
-- instead: one incident, one action, one day.
|
||||
CREATE TABLE incident_ack_tokens (
|
||||
token_hash TEXT PRIMARY KEY, -- SHA-256 of the raw token, as with api_keys
|
||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
created_at BIGINT NOT NULL,
|
||||
expires_at BIGINT NOT NULL
|
||||
);
|
||||
|
||||
CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens(expires_at);
|
||||
@@ -1,25 +0,0 @@
|
||||
-- A system administrator role, and the first thing in this server that one user
|
||||
-- can do and another cannot.
|
||||
--
|
||||
-- Until now every authenticated caller could create and delete users, set
|
||||
-- anybody's password and mint API keys for anybody — auth.go said so in a
|
||||
-- comment. That was defensible with one operator and a hand-made account; it is
|
||||
-- not once people sign themselves up (see #7).
|
||||
--
|
||||
-- EVERY EXISTING USER BECOMES AN ADMIN. They already hold these powers, so
|
||||
-- this migration changes nobody's access: it names what is already true, and
|
||||
-- leaves demotion as a deliberate act somebody performs afterwards. The
|
||||
-- alternative — promoting only user 1 — would silently strip the others, and
|
||||
-- could leave an install whose only admin is an account nobody has a password
|
||||
-- for.
|
||||
--
|
||||
-- New users are not admins: the column defaults to false, and the only ways to
|
||||
-- become one are this backfill, the bootstrap endpoint, or an existing admin
|
||||
-- granting it.
|
||||
ALTER TABLE users ADD COLUMN is_admin BOOLEAN NOT NULL DEFAULT false;
|
||||
|
||||
UPDATE users SET is_admin = true;
|
||||
|
||||
-- The queue's assignment dropdown and the on-call schedule read every user, and
|
||||
-- the admin screens in #5 will filter on this.
|
||||
CREATE INDEX users_is_admin_idx ON users(is_admin) WHERE is_admin;
|
||||
@@ -1,103 +0,0 @@
|
||||
-- Teams: the unit of tenancy. Everything a person works on now belongs to one.
|
||||
--
|
||||
-- Until this migration the install was one shared space — every user saw every
|
||||
-- alert and every incident, and the Alertmanager webhook was unauthenticated, so
|
||||
-- anything that could reach the port could open an incident for everybody.
|
||||
--
|
||||
-- The shape, in one paragraph: a team owns its incidents, alerts, schedule and
|
||||
-- integrations. A user belongs to as many teams as they like, with a role in
|
||||
-- each: an `owner` configures the team, a `member` works its incidents. An
|
||||
-- integration key is what an alert arrives on, and the key is what says which
|
||||
-- team the alert belongs to.
|
||||
--
|
||||
-- EVERYTHING EXISTING MOVES INTO ONE DEFAULT TEAM, and every existing user
|
||||
-- becomes an owner of it. That keeps an upgrade a no-op for the people using it:
|
||||
-- the same queue, the same schedule, the same incidents, with a name on them.
|
||||
|
||||
CREATE TABLE teams (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
-- role is free text with a CHECK rather than an enum, so adding a third role
|
||||
-- later is a migration and not a type rewrite.
|
||||
CREATE TABLE team_members (
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
|
||||
joined_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
PRIMARY KEY (team_id, user_id)
|
||||
);
|
||||
|
||||
CREATE INDEX team_members_user_idx ON team_members(user_id);
|
||||
|
||||
-- How alerts get in, and the only thing that says which team they belong to.
|
||||
-- The key is stored as a SHA-256 hash, like api_keys and the ack tokens: a
|
||||
-- leaked database gives nobody the ability to post alerts.
|
||||
CREATE TABLE integrations (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
kind TEXT NOT NULL CHECK (kind IN ('alertmanager')),
|
||||
name TEXT NOT NULL,
|
||||
key_hash TEXT NOT NULL UNIQUE,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
last_used_at BIGINT
|
||||
);
|
||||
|
||||
CREATE INDEX integrations_team_idx ON integrations(team_id);
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- The default team, and everything that already exists moving into it.
|
||||
--
|
||||
-- Created unconditionally, even on an empty install, so there is always a team
|
||||
-- for the bootstrap user to land in and for the first integration to hang off.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
INSERT INTO teams (name) VALUES ('Default');
|
||||
|
||||
INSERT INTO team_members (team_id, user_id, role)
|
||||
SELECT (SELECT id FROM teams WHERE name = 'Default'), id, 'owner' FROM users;
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- team_id on everything a team owns.
|
||||
--
|
||||
-- Added nullable, backfilled, then made NOT NULL: adding a NOT NULL column with
|
||||
-- no default to a table with rows is rejected, and a DEFAULT pointing at the
|
||||
-- default team would quietly keep working after the default team is gone.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
ALTER TABLE alerts ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
||||
ALTER TABLE incidents ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
||||
ALTER TABLE schedule_entries ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
||||
|
||||
UPDATE alerts SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
||||
UPDATE incidents SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
||||
UPDATE schedule_entries SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
||||
|
||||
ALTER TABLE alerts ALTER COLUMN team_id SET NOT NULL;
|
||||
ALTER TABLE incidents ALTER COLUMN team_id SET NOT NULL;
|
||||
ALTER TABLE schedule_entries ALTER COLUMN team_id SET NOT NULL;
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- The uniqueness rules were all written for one tenant, and every one of them
|
||||
-- is wrong now: two teams monitoring two clusters legitimately see the same
|
||||
-- fingerprint, the same groupKey, and want somebody on call on the same day.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
ALTER TABLE alerts DROP CONSTRAINT alerts_fingerprint_key;
|
||||
CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts(team_id, fingerprint);
|
||||
|
||||
DROP INDEX incidents_open_group_key_idx;
|
||||
-- Still load-bearing, now per team: at most one OPEN incident per group_key
|
||||
-- within a team. This is what makes "resolved incident + a new alert occurrence
|
||||
-- = a new incident" work, and what the webhook's find-or-open lookup relies on.
|
||||
CREATE UNIQUE INDEX incidents_open_group_key_idx
|
||||
ON incidents(team_id, group_key) WHERE resolved_at IS NULL;
|
||||
|
||||
ALTER TABLE schedule_entries DROP CONSTRAINT schedule_entries_date_key;
|
||||
CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries(team_id, date);
|
||||
|
||||
-- The list views all filter by team first.
|
||||
CREATE INDEX alerts_team_received_idx ON alerts(team_id, received_at DESC);
|
||||
CREATE INDEX incidents_team_triggered_idx ON incidents(team_id, triggered_at DESC);
|
||||
@@ -1,39 +0,0 @@
|
||||
-- Dead man's switches become a team's own configuration.
|
||||
--
|
||||
-- They were three environment variables — TERDUT_DEADMAN_MATCHERS, _TIMEOUT and
|
||||
-- _SEVERITY — which made them one setting for the whole install. That was the
|
||||
-- last piece of the alerting path a team could not control: a team could take
|
||||
-- its own alerts on its own key and still not say which of them were
|
||||
-- heartbeats, or how long a silence had to last before somebody was paged.
|
||||
--
|
||||
-- One row per team rather than one row per switch. The unit of monitoring is
|
||||
-- still the fingerprint, as it always was — two clusters sending the same
|
||||
-- heartbeat alertname are two independent switches — and the matcher string
|
||||
-- keeps the format the environment variable used, so a value can be moved from
|
||||
-- one to the other unchanged.
|
||||
--
|
||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
||||
-- inserts a row per team at startup from its own configuration, and the same
|
||||
-- values therefore carry forward into the first team's row without anybody
|
||||
-- retyping them. See seedDeadmanConfigs.
|
||||
CREATE TABLE deadman_configs (
|
||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
||||
|
||||
-- ";" separates matchers, "," the label conditions within one, "=" is exact
|
||||
-- equality: `alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat`.
|
||||
-- Every matcher must name an alertname. Empty watches nothing.
|
||||
matchers TEXT NOT NULL DEFAULT '',
|
||||
|
||||
-- Seconds rather than a Go duration string: the column is compared and
|
||||
-- arithmetic is done on it, and a value that has to be parsed before it can
|
||||
-- be believed is a value that can be stored unparseable. Zero disables the
|
||||
-- team's switches entirely.
|
||||
timeout_seconds BIGINT NOT NULL DEFAULT 0,
|
||||
|
||||
-- The severity these incidents open at. They have no member alerts to
|
||||
-- derive one from, and a heartbeat's own severity label is meaningless —
|
||||
-- Watchdog ships as "none".
|
||||
severity TEXT NOT NULL DEFAULT 'critical',
|
||||
|
||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
@@ -1,35 +0,0 @@
|
||||
-- Settings that an administrator can change without a redeploy, and the flag
|
||||
-- that takes an account out of use without deleting it.
|
||||
--
|
||||
-- Three of the server's tunables were environment variables, which meant
|
||||
-- changing how long an incident waits before it is paged again required editing
|
||||
-- a chart, merging it, and waiting for a reconcile. They are behaviour, not
|
||||
-- infrastructure, and the difference is who needs to change them and how often.
|
||||
--
|
||||
-- What stays in the environment: the ntfy URL and token, the database DSN, the
|
||||
-- listen address and the public URL. Those are where the server is plugged in
|
||||
-- rather than how it behaves, they are needed before the database is open, and
|
||||
-- two of them are credentials.
|
||||
--
|
||||
-- Key/value rather than a column per setting. A settings table with one row and
|
||||
-- a column per knob needs a migration for every new knob, and #6 and #7 will
|
||||
-- both add some. The cost is that values are text and the accessor has to say
|
||||
-- what type it wanted; settings.go does that in one place.
|
||||
--
|
||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
||||
-- inserts each key from its own configuration at startup, once, so an install
|
||||
-- that upgrades keeps exactly the behaviour it had. See SeedSettings.
|
||||
CREATE TABLE settings (
|
||||
key TEXT PRIMARY KEY,
|
||||
value TEXT NOT NULL,
|
||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
-- Disabling an account rather than deleting it: the person has left, or the
|
||||
-- credential is suspect, and their incidents, acknowledgements and timeline
|
||||
-- entries must stay exactly where they are. Deleting a user nulls their
|
||||
-- acknowledged_by and assigned_to, which quietly rewrites history.
|
||||
--
|
||||
-- A disabled user cannot sign in and their API keys stop working, but they are
|
||||
-- still a name the timeline can show and still a member of their teams.
|
||||
ALTER TABLE users ADD COLUMN disabled_at BIGINT;
|
||||
@@ -1,95 +0,0 @@
|
||||
-- Escalation: page somebody else when the first person does not answer.
|
||||
--
|
||||
-- This is the gap the whole multi-tenancy line of work was opened to close.
|
||||
-- Until now an unacknowledged incident re-paged the same topic every
|
||||
-- notify_repeat forever, which is a louder version of the same silence: if the
|
||||
-- person on call is asleep, has no signal, or has left, nothing else happens.
|
||||
--
|
||||
-- Shape: one policy per team, an ordered list of levels, each level with a
|
||||
-- timeout and a set of targets. When a level's timeout passes and the incident
|
||||
-- is still triggered, the next level is paged. When the last level passes, the
|
||||
-- chain repeats repeat_count times, and then the team's fallback topic is paged
|
||||
-- once as the end of the line.
|
||||
--
|
||||
-- A team WITHOUT a policy keeps exactly today's behaviour: page the assignee,
|
||||
-- then remind on the same topic. Escalation is opt-in per team, and the two
|
||||
-- never both run for one incident -- see enqueueReminders.
|
||||
CREATE TABLE escalation_policies (
|
||||
-- One per team for now, hence the team as the key rather than an id with a
|
||||
-- unique index: routing different alerts to different chains needs the
|
||||
-- alert to carry something to route ON, which is a separate question.
|
||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
||||
|
||||
-- How many extra times to run the whole chain after it has been walked
|
||||
-- once. 0 means walk it once and stop at the fallback.
|
||||
repeat_count BIGINT NOT NULL DEFAULT 0 CHECK (repeat_count >= 0 AND repeat_count <= 10),
|
||||
|
||||
-- Where the last page goes when every level has been tried. Per team now:
|
||||
-- TERDUT_NTFY_FALLBACK_TOPIC was one topic for the whole install, which in
|
||||
-- a multi-team server pages the wrong people. Empty means the chain simply
|
||||
-- ends.
|
||||
fallback_topic TEXT NOT NULL DEFAULT '',
|
||||
|
||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
CREATE TABLE escalation_levels (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
team_id BIGINT NOT NULL REFERENCES escalation_policies(team_id) ON DELETE CASCADE,
|
||||
-- 1-based, dense. The API rewrites the whole ladder on every edit rather
|
||||
-- than patching one rung, so there is no way to leave a gap.
|
||||
position BIGINT NOT NULL,
|
||||
-- How long this level has to produce an acknowledgement before the next one
|
||||
-- is paged. Seconds, like every other duration in this schema.
|
||||
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
|
||||
|
||||
UNIQUE (team_id, position)
|
||||
);
|
||||
|
||||
-- Who a level pages. Either a named person, or whoever the team's rota says is
|
||||
-- on call today -- which is the target that keeps working when the rota
|
||||
-- changes and nobody remembers to edit the policy.
|
||||
CREATE TABLE escalation_targets (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
level_id BIGINT NOT NULL REFERENCES escalation_levels(id) ON DELETE CASCADE,
|
||||
kind TEXT NOT NULL CHECK (kind IN ('user', 'oncall')),
|
||||
-- Set for kind='user', NULL for kind='oncall'.
|
||||
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
|
||||
|
||||
CHECK ((kind = 'user' AND user_id IS NOT NULL) OR (kind = 'oncall' AND user_id IS NULL))
|
||||
);
|
||||
|
||||
CREATE INDEX escalation_targets_level_idx ON escalation_targets(level_id);
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- Where an incident is in its chain.
|
||||
--
|
||||
-- On the incident rather than in a side table: it is read on every notifier
|
||||
-- tick alongside the incident's status, and one row per incident is exactly
|
||||
-- what the state is.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
-- 0 means no level has been paged yet, which is the state of every incident
|
||||
-- that existed before escalation and of every incident in a team with no
|
||||
-- policy. 1 is the first level.
|
||||
ALTER TABLE incidents ADD COLUMN escalation_level BIGINT NOT NULL DEFAULT 0;
|
||||
|
||||
-- When the current level was entered, and therefore what its timeout is
|
||||
-- measured from. NULL while escalation_level is 0.
|
||||
ALTER TABLE incidents ADD COLUMN escalation_level_at BIGINT;
|
||||
|
||||
-- How many times the chain has been walked in full. Compared against the
|
||||
-- policy's repeat_count.
|
||||
ALTER TABLE incidents ADD COLUMN escalation_round BIGINT NOT NULL DEFAULT 0;
|
||||
|
||||
-- The notifier's escalation query: incidents still waiting, oldest level first.
|
||||
CREATE INDEX incidents_escalation_idx
|
||||
ON incidents(escalation_level_at)
|
||||
WHERE resolved_at IS NULL AND status = 'triggered';
|
||||
|
||||
-- 'escalated' joins the outbox kinds: a page that went out because nobody
|
||||
-- answered the last one, which is worth telling apart from the first page and
|
||||
-- from a reminder when reading the timeline or debugging a delivery.
|
||||
ALTER TABLE notifications DROP CONSTRAINT notifications_kind_check;
|
||||
ALTER TABLE notifications ADD CONSTRAINT notifications_kind_check
|
||||
CHECK (kind IN ('triggered', 'reminder', 'resolved', 'escalated'));
|
||||
@@ -1,49 +0,0 @@
|
||||
-- Self-service sign-up, and the invite links that make it useful.
|
||||
--
|
||||
-- Until now the only way to get an account was for somebody who already had one
|
||||
-- to create it, and the login page told people to "ask an admin". That is a
|
||||
-- workable arrangement for one operator and an impossible one for a team.
|
||||
--
|
||||
-- An invite is a link, not an email: this server has no SMTP and adding it to
|
||||
-- send one message would be a new subsystem to run, secure and monitor. The
|
||||
-- person inviting sends the link however they already talk to the person they
|
||||
-- are inviting.
|
||||
CREATE TABLE invites (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
|
||||
-- SHA-256 of the raw token, like api_keys, the integration keys and the
|
||||
-- acknowledgement tokens. A leaked database hands nobody an account.
|
||||
token_hash TEXT NOT NULL UNIQUE,
|
||||
|
||||
-- Which team the invitee lands in, and as what. An invite always names a
|
||||
-- team: an account in no team sees an empty queue and can be paged by
|
||||
-- nobody, which is not a state to invite somebody into.
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
|
||||
|
||||
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
|
||||
-- Invites expire. A link that works forever is a credential nobody
|
||||
-- remembers issuing, sitting in a chat log.
|
||||
expires_at BIGINT NOT NULL,
|
||||
|
||||
-- Single-use by default: max_uses 1. A team onboarding six people at once
|
||||
-- can raise it rather than minting six links.
|
||||
max_uses BIGINT NOT NULL DEFAULT 1 CHECK (max_uses > 0 AND max_uses <= 100),
|
||||
uses BIGINT NOT NULL DEFAULT 0,
|
||||
|
||||
-- Revoked by hand, separately from expiry, so "this link is no longer
|
||||
-- wanted" and "this link timed out" stay distinguishable in the listing.
|
||||
revoked_at BIGINT
|
||||
);
|
||||
|
||||
CREATE INDEX invites_team_idx ON invites(team_id);
|
||||
|
||||
-- Who redeemed which invite. Kept after the invite is gone — the answer to "how
|
||||
-- did this account get here" should outlive the link that made it.
|
||||
ALTER TABLE users ADD COLUMN invited_via BIGINT REFERENCES invites(id) ON DELETE SET NULL;
|
||||
|
||||
-- Where a person is in the first-run checklist, so it can be resumed and
|
||||
-- dismissed rather than nagging forever. One row per user, created on demand.
|
||||
ALTER TABLE users ADD COLUMN onboarding_dismissed_at BIGINT;
|
||||
@@ -1,23 +0,0 @@
|
||||
-- Similar incidents: a signature per incident, so "has this happened before"
|
||||
-- is an indexed equality instead of a search.
|
||||
--
|
||||
-- The signature is the alert name plus the group labels that identify WHAT is
|
||||
-- broken, minus the ones that only say WHERE it happened to run this time
|
||||
-- (instance, pod, ...). Two incidents with the same signature in the same team
|
||||
-- are the same problem for a responder's purposes.
|
||||
--
|
||||
-- Computed in Go for new incidents (incidentSignature in incident_store.go).
|
||||
-- The backfill below MUST produce the same string; keep the volatile list in
|
||||
-- both places in step.
|
||||
ALTER TABLE incidents ADD COLUMN signature TEXT NOT NULL DEFAULT '';
|
||||
|
||||
UPDATE incidents SET signature =
|
||||
COALESCE(NULLIF(group_labels->>'alertname', ''), title) || '|' ||
|
||||
COALESCE((
|
||||
SELECT string_agg(e.k || '=' || e.v, ',' ORDER BY e.k)
|
||||
FROM jsonb_each_text(incidents.group_labels) AS e(k, v)
|
||||
WHERE e.k <> 'alertname'
|
||||
AND e.k NOT IN ('instance', 'pod', 'pod_name', 'pod_ip', 'container', 'container_name', 'endpoint')
|
||||
), '');
|
||||
|
||||
CREATE INDEX incidents_signature_idx ON incidents(team_id, signature, triggered_at DESC);
|
||||
@@ -1,54 +0,0 @@
|
||||
-- Dead man's switches become rows of their own.
|
||||
--
|
||||
-- 004 kept a team's switches in one string with one timeout and one severity,
|
||||
-- which was enough to configure them and not enough to show them: there was no
|
||||
-- thing to list, nothing to hang a status on, and every switch in a team had to
|
||||
-- share a deadline. A row per switch gives each its own name, matcher, timeout
|
||||
-- and severity, and gives the Team → Switches page something to be a list of.
|
||||
--
|
||||
-- The matcher keeps the syntax the string used, one matcher per row:
|
||||
-- `alertname=Watchdog,cluster=prod`. The unit of monitoring is still the
|
||||
-- fingerprint, so a matcher that many clusters satisfy is still one switch row
|
||||
-- watching several independent heartbeats.
|
||||
CREATE TABLE deadman_switches (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
|
||||
-- What the owner calls it. Defaults to the matcher when they do not say.
|
||||
name TEXT NOT NULL,
|
||||
|
||||
-- "," separates the label conditions, "=" is exact equality, and alertname is
|
||||
-- mandatory: it is what keeps the sweeper's candidate query on an index.
|
||||
matcher TEXT NOT NULL,
|
||||
|
||||
-- Seconds of silence before the switch is declared dead. Never zero: a switch
|
||||
-- that cannot fire is deleted, not disabled.
|
||||
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
|
||||
|
||||
-- The severity its incidents open at. See 004 for why they carry their own.
|
||||
severity TEXT NOT NULL DEFAULT 'critical',
|
||||
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
CREATE INDEX deadman_switches_team_idx ON deadman_switches (team_id);
|
||||
|
||||
-- Carry every team's configuration over, one row per matcher. A team whose
|
||||
-- timeout was zero had switches turned off, which is now "no rows".
|
||||
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
||||
SELECT c.team_id, btrim(m), btrim(m), c.timeout_seconds, c.severity
|
||||
FROM deadman_configs c,
|
||||
LATERAL regexp_split_to_table(c.matchers, ';') AS m
|
||||
WHERE c.timeout_seconds > 0
|
||||
AND btrim(m) <> ''
|
||||
ORDER BY c.team_id;
|
||||
|
||||
-- The server seeds environment defaults into teams once, and remembers that it
|
||||
-- did. An install that had a row per team was already seeded; without this
|
||||
-- marker the first start after upgrading would seed teams that had switched
|
||||
-- theirs off.
|
||||
INSERT INTO settings (key, value)
|
||||
SELECT 'deadman_seeded', '1'
|
||||
WHERE EXISTS (SELECT 1 FROM deadman_configs);
|
||||
|
||||
DROP TABLE deadman_configs;
|
||||
@@ -1,21 +0,0 @@
|
||||
-- Which alert source an alert last arrived on.
|
||||
--
|
||||
-- Team -> Sources shows when each source last posted, which integrations
|
||||
-- already knew (last_used_at, stamped on every webhook). What it could not say
|
||||
-- was what a source delivered: an alert never recorded the key it came in on, so
|
||||
-- "prod alertmanager" and "staging alertmanager" were indistinguishable once
|
||||
-- inside. This column is that link, and lets the page show each source's last
|
||||
-- alert and how many alerts it has kept fresh over the past day.
|
||||
--
|
||||
-- Last sender wins: every accepted payload restamps it, the way it advances
|
||||
-- received_at. Two sources posting the same fingerprint into one team is
|
||||
-- already one alert, and it is attributed to whichever spoke last.
|
||||
--
|
||||
-- Nullable, and not backfilled. Alerts that arrived before this migration have
|
||||
-- no source, and NULL says so honestly rather than guessing. It heals by itself:
|
||||
-- Alertmanager re-sends every alert each repeat_interval, and each re-send is an
|
||||
-- accepted payload. Deleting a source keeps its alerts, unattributed.
|
||||
ALTER TABLE alerts ADD COLUMN integration_id BIGINT REFERENCES integrations(id) ON DELETE SET NULL;
|
||||
|
||||
CREATE INDEX alerts_integration_idx ON alerts (integration_id, received_at)
|
||||
WHERE integration_id IS NOT NULL;
|
||||
@@ -1,60 +0,0 @@
|
||||
-- Single sign-on through an OpenID Connect provider (Authentik, and anything
|
||||
-- else that speaks OIDC).
|
||||
--
|
||||
-- Four things change, and none of them touches a password user: every new column
|
||||
-- has a default that says "this is how it has always worked".
|
||||
--
|
||||
-- 1. user_identities says which provider account a user is. It is keyed on
|
||||
-- (issuer, subject), never on email or username: those are mutable at the
|
||||
-- provider, and a recycled address must not inherit somebody's account. A
|
||||
-- user can have several identities (a second provider later), and none at all
|
||||
-- (a local, password-only user), which is why this is a table and not two
|
||||
-- columns on users.
|
||||
--
|
||||
-- 2. team_members.source and users.admin_source record who granted a role. 'oidc'
|
||||
-- rows are owned by the group sync: it adds them when a group grants access
|
||||
-- and removes them when it stops, and nothing else may edit them. 'manual' rows
|
||||
-- are everything that existed before this migration, and are never touched by
|
||||
-- the sync. Without the marker the sync could not tell a membership it created
|
||||
-- from one an owner added by hand, and would have to either leave stale access
|
||||
-- behind or delete people it had no business deleting.
|
||||
--
|
||||
-- 3. sessions.max_expires_at is a hard ceiling on a session's life. Ordinary
|
||||
-- sessions slide for as long as they are used; a session made by an SSO login
|
||||
-- must not, because the login is the only moment the groups are re-read.
|
||||
-- Capping the session is what makes "removed from the group in the provider"
|
||||
-- take effect within a bounded time. NULL means no ceiling.
|
||||
--
|
||||
-- 4. oidc_logins holds a login that has been started and not yet finished: the
|
||||
-- state, nonce and PKCE verifier the callback must see again. A row rather
|
||||
-- than a signed cookie, so it survives a restart and needs no signing key.
|
||||
-- Only the hash of the state is stored, like every other token here; the
|
||||
-- nonce and verifier are useless without the state that names the row.
|
||||
CREATE TABLE user_identities (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
issuer TEXT NOT NULL,
|
||||
subject TEXT NOT NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
last_login_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
UNIQUE (issuer, subject)
|
||||
);
|
||||
|
||||
CREATE INDEX user_identities_user_idx ON user_identities (user_id);
|
||||
|
||||
ALTER TABLE team_members
|
||||
ADD COLUMN source TEXT NOT NULL DEFAULT 'manual' CHECK (source IN ('manual', 'oidc'));
|
||||
|
||||
ALTER TABLE users
|
||||
ADD COLUMN admin_source TEXT NOT NULL DEFAULT 'manual' CHECK (admin_source IN ('manual', 'oidc'));
|
||||
|
||||
ALTER TABLE sessions ADD COLUMN max_expires_at BIGINT;
|
||||
|
||||
CREATE TABLE oidc_logins (
|
||||
state_hash TEXT PRIMARY KEY,
|
||||
nonce TEXT NOT NULL,
|
||||
pkce_verifier TEXT NOT NULL,
|
||||
expires_at BIGINT NOT NULL
|
||||
);
|
||||
|
||||
CREATE INDEX oidc_logins_expires_idx ON oidc_logins (expires_at);
|
||||
@@ -1,40 +0,0 @@
|
||||
-- Signing in from a terminal, for clients that cannot open a browser on the
|
||||
-- machine they run on (the TUI over SSH is the reason).
|
||||
--
|
||||
-- The flow is the OAuth device authorization grant, run by terdut itself rather
|
||||
-- than the identity provider, so the terminal never talks to the provider and
|
||||
-- the server issues its ordinary session at the end:
|
||||
--
|
||||
-- 1. The terminal asks for a login and gets two secrets: a device code it
|
||||
-- keeps and polls with, and a short user code it shows the person.
|
||||
-- 2. The person opens the verification URL on any device, signs in by whatever
|
||||
-- means the server offers, sees the user code, and approves it.
|
||||
-- 3. The terminal's next poll finds the row approved and is given a session.
|
||||
--
|
||||
-- Only the hash of the device code is stored, like every other token here: the
|
||||
-- device code is what earns a session, so a database read must not yield one.
|
||||
-- The user code is shown on screens and typed by people, so it is stored as is;
|
||||
-- on its own it can only be approved, never redeemed.
|
||||
--
|
||||
-- user_id is the person who approved. It is empty until then, and the session
|
||||
-- is minted at redemption, not at approval: an approval nobody collects must not
|
||||
-- leave a live session lying about.
|
||||
--
|
||||
-- last_polled_at lets the server refuse a client that polls faster than the
|
||||
-- interval it was told.
|
||||
CREATE TABLE device_logins (
|
||||
device_hash TEXT PRIMARY KEY,
|
||||
user_code TEXT NOT NULL UNIQUE,
|
||||
status TEXT NOT NULL DEFAULT 'pending' CHECK (status IN ('pending', 'approved', 'denied')),
|
||||
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
|
||||
expires_at BIGINT NOT NULL,
|
||||
last_polled_at BIGINT NOT NULL DEFAULT 0
|
||||
);
|
||||
|
||||
CREATE INDEX device_logins_expires_idx ON device_logins (expires_at);
|
||||
|
||||
-- Where to send the browser once a single sign-on login completes. A person who
|
||||
-- opens /device?code=... without a session has to sign in first and then come
|
||||
-- back to it, and the same is true of any other deep link. Validated when it is
|
||||
-- stored: only a path on this server is ever kept.
|
||||
ALTER TABLE oidc_logins ADD COLUMN next TEXT NOT NULL DEFAULT '/';
|
||||
@@ -1,26 +0,0 @@
|
||||
-- Per-team OIDC group configuration, replacing the global
|
||||
-- TERDUT_OIDC_GROUP_MAPPINGS env var.
|
||||
--
|
||||
-- Group -> team -> role used to be one global list an operator set for the
|
||||
-- whole install, matched against a team by name, and the sync would create
|
||||
-- the team if no team by that name existed yet. That put the decision of
|
||||
-- which group controls a team in the server's environment rather than the
|
||||
-- team's own hands, meant changing it needed an env var edit and a restart,
|
||||
-- and let a typo in a team name silently create a stray team.
|
||||
--
|
||||
-- Each team now names, itself, which group grants membership and which
|
||||
-- grants ownership. Nullable: most teams need neither. No uniqueness
|
||||
-- constraint on either column — two teams may legitimately watch the same
|
||||
-- provider group (a broad team and a narrower one both keyed off overlapping
|
||||
-- groups is a choice for their owners to make, not one the schema should
|
||||
-- refuse).
|
||||
--
|
||||
-- BREAKING CHANGE, deliberately not auto-migrated: TERDUT_OIDC_GROUP_MAPPINGS
|
||||
-- stops being read as of this version, and the sync no longer creates a team
|
||||
-- by name. Every team's group binding must be set again through
|
||||
-- PUT /api/teams/{teamID}/oidc-groups. Until an owner does that, an
|
||||
-- OIDC-sourced membership in that team is dropped at that user's next SSO
|
||||
-- sign-in, the same way any other loss of group access is handled. See the
|
||||
-- README's OIDC section.
|
||||
ALTER TABLE teams ADD COLUMN oidc_member_group TEXT;
|
||||
ALTER TABLE teams ADD COLUMN oidc_owner_group TEXT;
|
||||
@@ -1,43 +0,0 @@
|
||||
-- Service accounts: a scoped, non-human credential for automation (e.g.
|
||||
-- terdut-operator) that needs to manage teams, escalation policies, dead
|
||||
-- man's switches, integrations and OIDC group bindings without impersonating
|
||||
-- a human user. See SERVICE-ACCOUNTS.md for the design this implements.
|
||||
--
|
||||
-- Deliberately not a users row: no password_hash, no is_admin, no
|
||||
-- user_identities linkage, so a service account can never be pulled into
|
||||
-- OIDC group sync or password login, and is never mistaken for a human in an
|
||||
-- audit trail.
|
||||
--
|
||||
-- scope is 'instance' (acts with the same reach system administration has
|
||||
-- over teams: create one, list them, mint a 'team'-scoped account against
|
||||
-- any of them) or 'team' (acts as that one team's owner, and nothing else).
|
||||
-- The CHECK ties team_id's presence to scope directly, rather than leaving it
|
||||
-- to application code to keep the two consistent.
|
||||
CREATE TABLE service_accounts (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
|
||||
team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
|
||||
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
CONSTRAINT service_accounts_scope_team_id_chk CHECK (
|
||||
(scope = 'team' AND team_id IS NOT NULL) OR
|
||||
(scope = 'instance' AND team_id IS NULL)
|
||||
)
|
||||
);
|
||||
|
||||
CREATE INDEX service_accounts_team_id_idx ON service_accounts(team_id);
|
||||
|
||||
-- One account, many keys: rotation is minting a new one and revoking the
|
||||
-- old, the same shape api_keys already has, so an account's identity and
|
||||
-- audit history survive a rotation instead of being recreated by it.
|
||||
CREATE TABLE service_account_keys (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
|
||||
key_hash TEXT NOT NULL UNIQUE,
|
||||
name TEXT NOT NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
last_used_at BIGINT
|
||||
);
|
||||
|
||||
CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys(service_account_id);
|
||||
@@ -1,39 +0,0 @@
|
||||
-- Service-account actors on incident mutations (terdut-server#25). A
|
||||
-- team-scoped service account acknowledging/resolving/snoozing/noting an
|
||||
-- incident is not a users row, so it cannot be written into
|
||||
-- acknowledged_by/incident_events.user_id — doing so either violates the
|
||||
-- users(id) FK (new rows) or, for incident_events.user_id, silently matches
|
||||
-- zero rows on delete. These columns are the service-account-shaped parallel
|
||||
-- to the existing human ones: nullable, mutually exclusive with their human
|
||||
-- counterpart, ON DELETE SET NULL so a deleted service account doesn't take
|
||||
-- the incident history with it.
|
||||
ALTER TABLE incidents
|
||||
ADD COLUMN acknowledged_by_service_account_id BIGINT
|
||||
REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||
|
||||
ALTER TABLE incident_events
|
||||
ADD COLUMN service_account_id BIGINT
|
||||
REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||
|
||||
-- At most one actor kind per row: both NULL ("the server acted") is valid,
|
||||
-- exactly one set is valid, both set is a bug this constraint refuses to
|
||||
-- store rather than silently accepting.
|
||||
ALTER TABLE incidents
|
||||
ADD CONSTRAINT incidents_ack_actor_xor_chk CHECK (
|
||||
acknowledged_by IS NULL OR acknowledged_by_service_account_id IS NULL
|
||||
);
|
||||
|
||||
ALTER TABLE incident_events
|
||||
ADD CONSTRAINT incident_events_actor_xor_chk CHECK (
|
||||
user_id IS NULL OR service_account_id IS NULL
|
||||
);
|
||||
|
||||
CREATE INDEX incidents_acknowledged_by_service_account_id_idx
|
||||
ON incidents(acknowledged_by_service_account_id);
|
||||
CREATE INDEX incident_events_service_account_id_idx
|
||||
ON incident_events(service_account_id);
|
||||
|
||||
-- assigned_to_service_account_id is deliberately not added here: it would sit
|
||||
-- unpopulated until handleIncidentAssign itself tracks an actor, which is a
|
||||
-- separate, pre-existing gap (it records the assignee today, never the
|
||||
-- actor, for humans either) tracked in its own follow-up issue.
|
||||
@@ -1,16 +0,0 @@
|
||||
-- Backs the rate limiters (failed logins, sign-ups, OIDC/device start) with
|
||||
-- Postgres instead of an in-memory map, now that the server runs more than
|
||||
-- one replica in production (v0.37.0): a counter that only ever sees its own
|
||||
-- pod's traffic quietly let every one of these limits through multiplied by
|
||||
-- the replica count.
|
||||
--
|
||||
-- window_start is the start of the current fixed window for key, in the same
|
||||
-- "unix seconds" shape every other timestamp in this schema uses. The window
|
||||
-- resets rather than slides, matching the in-memory limiter it replaces:
|
||||
-- once a key's window is older than the limiter's window length, the next
|
||||
-- failure starts a fresh one instead of extending the stale one.
|
||||
CREATE TABLE rate_limit_counters (
|
||||
key TEXT PRIMARY KEY,
|
||||
window_start BIGINT NOT NULL,
|
||||
count INT NOT NULL
|
||||
);
|
||||
@@ -1,7 +0,0 @@
|
||||
-- Optional expiry on a user's own API keys. NULL (the existing default for
|
||||
-- every row already in this table) means "never expires" -- the same
|
||||
-- behavior these keys have always had, so no existing integration breaks.
|
||||
-- Service account keys are deliberately NOT touched: they are a different
|
||||
-- table, managed by automation, and already distinguished by their own
|
||||
-- "tdsa_" prefix.
|
||||
ALTER TABLE api_keys ADD COLUMN expires_at BIGINT;
|
||||
@@ -1,21 +0,0 @@
|
||||
-- Who performed an assignment (terdut-server#35). On an 'assigned' event
|
||||
-- incident_events.user_id is the assignee, so the actor needs columns of its
|
||||
-- own. Only populated for 'assigned' events; every other event type keeps
|
||||
-- using user_id/service_account_id for the actor. Older 'assigned' rows stay
|
||||
-- NULL (the actor was never recorded). Same shape as migration 015: nullable,
|
||||
-- mutually exclusive, ON DELETE SET NULL.
|
||||
--
|
||||
-- assigned_to_service_account_id is still deliberately not added: making
|
||||
-- service accounts assignable is a separate change (request body, assignee
|
||||
-- picker, notifier, filters).
|
||||
ALTER TABLE incident_events
|
||||
ADD COLUMN actor_user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
ADD COLUMN actor_service_account_id BIGINT REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||
|
||||
ALTER TABLE incident_events
|
||||
ADD CONSTRAINT incident_events_assign_actor_xor_chk CHECK (
|
||||
actor_user_id IS NULL OR actor_service_account_id IS NULL
|
||||
);
|
||||
|
||||
CREATE INDEX incident_events_actor_user_id_idx ON incident_events(actor_user_id);
|
||||
CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events(actor_service_account_id);
|
||||
Reference in New Issue
Block a user