a4fbd60441
The core of #4, and what #1 is for: terdut stops being one shared space. A team owns its incidents, alerts, schedule and integrations; a user sees exactly the teams they are in. Everything that existed moves into one Default team and every existing user becomes an owner of it, so the upgrade is a no-op for the people using it. Ingestion is the load-bearing half. An alert arrives on a team's integration key, and the key is both the credential and the routing: it says that the sender may post, and which team the alerts belong to. That also closes the unauthenticated webhook -- the old path stays for one release, deprecated and routed to the oldest team, so an upgrade does not stop delivering while somebody edits the Alertmanager config. Scoping is enforced in as few places as possible, because the failure mode is silent. serveAs loads the caller's memberships once; list queries carry `team_id = ANY(...)`; and every incident route goes through incidentIDParam, which now parses the id AND checks the team in the same call, so a new handler cannot remember the first half and forget the second. Anything in another team is 404, never 403: whether an incident exists is that team's business. Two bugs this found, both of which would have been silent: * upsertAlerts decided "is this a new occurrence" by looking up the fingerprint alone. Across teams that made team B's first alert look like a re-send of team A's, so it opened no incident at all. The lookups are keyed on (team_id, fingerprint) now, as the index is. * Every uniqueness rule was written for one tenant. Two teams watching two clusters legitimately see the same fingerprint, the same groupKey, and want somebody on call on the same day; all three constraints move to include team_id. Roles inside a team are separate from the system administrator flag: an owner configures the team, a member works its incidents, and an admin is NOT implicitly in every team -- administration is about accounts, not about reading other people's incidents. An admin can still repair a team whose owner has left, which is why requireTeamOwner lets them through. A shift can only be given to somebody in the team. Paging a person who cannot open the incident is worse than paging nobody. The UI is updated only as far as keeping it working: it loads the viewer's teams with the session and uses the first one, since nobody has a second yet. "On call now" shows every team the viewer is in, named only when there is more than one, so the common case reads exactly as before. The team switcher, badges and per-team settings pages are the next step. Breaking for API clients: the schedule endpoints moved under the team, and /api/schedule/current returns an array rather than an object or a 404. terdut-tui will need a version for that. Per-team dead-man configuration is deliberately not here. A heartbeat's incident already opens in the team whose key received it, which is the part that matters for isolation; moving the matchers out of env into per-team rows is a change to how deadman.go is configured rather than to who sees what. Claude-Session: https://claude.ai/code/session_01RHPj4ggeFdEjKKfm4SHbD7
307 lines
10 KiB
Go
307 lines
10 KiB
Go
package api
|
|
|
|
import (
|
|
"context"
|
|
"database/sql"
|
|
"encoding/json"
|
|
"sort"
|
|
"strings"
|
|
"time"
|
|
|
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
|
)
|
|
|
|
// Values for incidents.resolution_source, recording who closed the incident:
|
|
// every member alert stopped firing, or a person decided it was done.
|
|
const (
|
|
incidentResolutionAlerts = "alerts"
|
|
incidentResolutionManual = "manual"
|
|
|
|
// incidentResolutionRecovered closes a dead man's switch incident whose
|
|
// heartbeat started arriving again. It cannot be "alerts": these incidents
|
|
// have no member alerts for the cascade to work from.
|
|
incidentResolutionRecovered = "recovered"
|
|
)
|
|
|
|
// Incident timeline event types. Stored as free text so adding one later is not
|
|
// a migration, but these are the ones the server writes.
|
|
const (
|
|
evTriggered = "triggered"
|
|
evAlertAdded = "alert_added"
|
|
evAlertResolved = "alert_resolved"
|
|
evAcknowledged = "acknowledged"
|
|
evUnacknowledged = "unacknowledged"
|
|
evAssigned = "assigned"
|
|
evSnoozed = "snoozed"
|
|
evUnsnoozed = "unsnoozed"
|
|
evResolved = "resolved"
|
|
evNote = "note"
|
|
evDeadmanSilent = "deadman_silent"
|
|
)
|
|
|
|
// severityLabel is the Alertmanager label an incident's severity is derived from.
|
|
const severityLabel = "severity"
|
|
|
|
// querier is satisfied by both *sql.DB and *sql.Tx, so the helpers below work
|
|
// inside the webhook's transaction and standalone from handlers and the sweeper.
|
|
type querier interface {
|
|
ExecContext(ctx context.Context, query string, args ...any) (sql.Result, error)
|
|
QueryContext(ctx context.Context, query string, args ...any) (*sql.Rows, error)
|
|
QueryRowContext(ctx context.Context, query string, args ...any) *sql.Row
|
|
}
|
|
|
|
const incidentSelectFrom = `
|
|
SELECT i.id, i.team_id, t.name, i.group_key, i.title, i.group_labels, i.status, i.severity,
|
|
i.triggered_at,
|
|
i.acknowledged_by, i.acknowledged_at, ack.username,
|
|
i.assigned_to, asg.username, i.snoozed_until,
|
|
i.resolved_at, i.resolution_source, i.archived_at
|
|
FROM incidents i
|
|
JOIN teams t ON t.id = i.team_id
|
|
LEFT JOIN users ack ON ack.id = i.acknowledged_by
|
|
LEFT JOIN users asg ON asg.id = i.assigned_to`
|
|
|
|
func scanIncident(s scanner) (models.Incident, error) {
|
|
var i models.Incident
|
|
var groupLabelsJSON string
|
|
var triggeredAt int64
|
|
var ackAt, snoozedUntil, resolvedAt, archivedAt *int64
|
|
|
|
if err := s.Scan(
|
|
&i.ID, &i.TeamID, &i.TeamName, &i.GroupKey, &i.Title, &groupLabelsJSON, &i.Status, &i.Severity,
|
|
&triggeredAt,
|
|
&i.AcknowledgedByID, &ackAt, &i.AcknowledgedByUser,
|
|
&i.AssignedToID, &i.AssignedToUser, &snoozedUntil,
|
|
&resolvedAt, &i.ResolutionSource, &archivedAt,
|
|
); err != nil {
|
|
return i, err
|
|
}
|
|
|
|
json.Unmarshal([]byte(groupLabelsJSON), &i.GroupLabels) //nolint:errcheck
|
|
i.TriggeredAt = time.Unix(triggeredAt, 0).UTC()
|
|
i.AcknowledgedAt = unixPtr(ackAt)
|
|
i.SnoozedUntil = unixPtr(snoozedUntil)
|
|
i.ResolvedAt = unixPtr(resolvedAt)
|
|
i.ArchivedAt = unixPtr(archivedAt)
|
|
return i, nil
|
|
}
|
|
|
|
// unixPtr converts a nullable Unix-second column to a nullable UTC time.
|
|
func unixPtr(sec *int64) *time.Time {
|
|
if sec == nil {
|
|
return nil
|
|
}
|
|
t := time.Unix(*sec, 0).UTC()
|
|
return &t
|
|
}
|
|
|
|
func fetchIncident(ctx context.Context, q querier, id int64) (models.Incident, error) {
|
|
return scanIncident(q.QueryRowContext(ctx, incidentSelectFrom+" WHERE i.id = $1", id))
|
|
}
|
|
|
|
// logEvent appends one entry to an incident's timeline. A nil userID means the
|
|
// server acted rather than a person.
|
|
func logEvent(ctx context.Context, q querier, incidentID int64, evType string, userID, alertID *int64, detail *string) error {
|
|
_, err := q.ExecContext(ctx, `
|
|
INSERT INTO incident_events (incident_id, type, user_id, alert_id, detail, created_at)
|
|
VALUES ($1, $2, $3, $4, $5, $6)`,
|
|
incidentID, evType, userID, alertID, detail, time.Now().Unix())
|
|
return err
|
|
}
|
|
|
|
// todayUTC is the schedule's day key. The schedule's smallest unit is one UTC day.
|
|
func todayUTC() string {
|
|
return time.Now().UTC().Format("2006-01-02")
|
|
}
|
|
|
|
// currentOnCall returns a team's on-call user for today, or nil when nobody is
|
|
// scheduled. A missing schedule entry is not an error — incidents just open
|
|
// unassigned.
|
|
//
|
|
// Per team: each team keeps its own rota, so two teams can have two different
|
|
// people on call on the same day, which was the point of scoping the schedule.
|
|
func currentOnCall(ctx context.Context, q querier, teamID int64) (*int64, error) {
|
|
var userID int64
|
|
err := q.QueryRowContext(ctx,
|
|
"SELECT user_id FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
|
teamID, todayUTC()).Scan(&userID)
|
|
if err == sql.ErrNoRows {
|
|
return nil, nil
|
|
}
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return &userID, nil
|
|
}
|
|
|
|
// severityRank orders the conventional Alertmanager severity label values.
|
|
// Anything unrecognised sorts below all of them rather than being dropped.
|
|
func severityRank(s string) int {
|
|
switch strings.ToLower(s) {
|
|
case "critical":
|
|
return 4
|
|
case "error":
|
|
return 3
|
|
case "warning":
|
|
return 2
|
|
case "info":
|
|
return 1
|
|
default:
|
|
return 0
|
|
}
|
|
}
|
|
|
|
// refreshSeverity raises an incident's severity to the highest `severity` label
|
|
// seen across its alerts.
|
|
//
|
|
// It is a high-water mark, never lowered: an incident that hit critical was a
|
|
// critical incident, even after the critical alert clears and a warning is all
|
|
// that is left firing. Downgrading a live incident would also quietly demote it
|
|
// in the queue while the work is still open.
|
|
func refreshSeverity(ctx context.Context, q querier, incidentID int64) error {
|
|
rows, err := q.QueryContext(ctx, `
|
|
SELECT a.labels ->> $1
|
|
FROM incident_alerts ia
|
|
JOIN alerts a ON a.id = ia.alert_id
|
|
WHERE ia.incident_id = $2`, severityLabel, incidentID)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
best := ""
|
|
for rows.Next() {
|
|
var sev *string
|
|
if err := rows.Scan(&sev); err != nil {
|
|
rows.Close()
|
|
return err
|
|
}
|
|
if sev != nil && severityRank(*sev) > severityRank(best) {
|
|
best = *sev
|
|
}
|
|
}
|
|
if err := rows.Err(); err != nil {
|
|
rows.Close()
|
|
return err
|
|
}
|
|
rows.Close()
|
|
|
|
if best == "" {
|
|
return nil
|
|
}
|
|
// The comparison lives in SQL so an unrelated concurrent update cannot be
|
|
// clobbered by a stale read.
|
|
_, err = q.ExecContext(ctx, `
|
|
UPDATE incidents SET severity = $1
|
|
WHERE id = $2
|
|
AND (severity IS NULL OR `+severityRankSQL("severity")+` < $3)`,
|
|
best, incidentID, severityRank(best))
|
|
return err
|
|
}
|
|
|
|
// severityRankSQL mirrors severityRank for use inside a statement. SQL cannot
|
|
// order these strings meaningfully on its own.
|
|
func severityRankSQL(col string) string {
|
|
return `CASE lower(COALESCE(` + col + `, ''))
|
|
WHEN 'critical' THEN 4
|
|
WHEN 'error' THEN 3
|
|
WHEN 'warning' THEN 2
|
|
WHEN 'info' THEN 1
|
|
ELSE 0 END`
|
|
}
|
|
|
|
// resolveIfSettled closes an incident once every alert under it has stopped
|
|
// firing — PagerDuty's cascade, and the only automatic route out of the open
|
|
// state. Reports whether it actually resolved anything.
|
|
func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, error) {
|
|
res, err := q.ExecContext(ctx, `
|
|
UPDATE incidents
|
|
SET status = 'resolved',
|
|
resolved_at = $1,
|
|
resolution_source = $2
|
|
WHERE id = $3
|
|
AND resolved_at IS NULL
|
|
-- An incident with no members yet is mid-creation, not settled.
|
|
AND EXISTS (SELECT 1 FROM incident_alerts ia WHERE ia.incident_id = incidents.id)
|
|
AND NOT EXISTS (SELECT 1
|
|
FROM incident_alerts ia
|
|
JOIN alerts a ON a.id = ia.alert_id
|
|
WHERE ia.incident_id = incidents.id
|
|
AND a.status = 'firing')`,
|
|
time.Now().Unix(), incidentResolutionAlerts, incidentID)
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
n, _ := res.RowsAffected()
|
|
if n == 0 {
|
|
return false, nil
|
|
}
|
|
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil); err != nil {
|
|
return false, err
|
|
}
|
|
// The all-clear goes only to whoever was paged in the first place, which
|
|
// enqueueResolved works out from the incident's own notification history.
|
|
// Manual resolution sends nothing: the person who closed it already knows.
|
|
return true, enqueueResolved(ctx, q, incidentID)
|
|
}
|
|
|
|
// acknowledgeIncident records that userID has picked an incident up, and reports
|
|
// whether it changed anything — an already-resolved incident is left alone.
|
|
// Shared by the authenticated handler and the Acknowledge button in a push
|
|
// notification, so both write the same state and the same timeline entry.
|
|
func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int64) (bool, error) {
|
|
res, err := q.ExecContext(ctx, `
|
|
UPDATE incidents
|
|
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_at = $2
|
|
WHERE id = $3 AND resolved_at IS NULL`,
|
|
userID, time.Now().Unix(), incidentID)
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
if n, _ := res.RowsAffected(); n == 0 {
|
|
return false, nil
|
|
}
|
|
return true, logEvent(ctx, q, incidentID, evAcknowledged, &userID, nil, nil)
|
|
}
|
|
|
|
// openIncidentForAlert returns the open incident an alert currently belongs to,
|
|
// or 0 when it has none. Used when an alert resolves or expires so the event
|
|
// lands on the right timeline.
|
|
func openIncidentForAlert(ctx context.Context, q querier, alertID int64) (int64, error) {
|
|
var id int64
|
|
err := q.QueryRowContext(ctx, `
|
|
SELECT i.id
|
|
FROM incident_alerts ia
|
|
JOIN incidents i ON i.id = ia.incident_id
|
|
WHERE ia.alert_id = $1 AND i.resolved_at IS NULL`, alertID).Scan(&id)
|
|
if err == sql.ErrNoRows {
|
|
return 0, nil
|
|
}
|
|
return id, err
|
|
}
|
|
|
|
// incidentTitle renders a human-readable title from Alertmanager's groupLabels,
|
|
// leading with the alert name and appending whatever else the operator grouped
|
|
// by. Falls back to the alert's own name when the payload carried no groupLabels.
|
|
func incidentTitle(groupLabels map[string]string, fallback string) string {
|
|
name := groupLabels["alertname"]
|
|
if name == "" {
|
|
name = fallback
|
|
}
|
|
if name == "" {
|
|
name = "Incident"
|
|
}
|
|
|
|
rest := make([]string, 0, len(groupLabels))
|
|
for k, v := range groupLabels {
|
|
if k == "alertname" {
|
|
continue
|
|
}
|
|
rest = append(rest, k+"="+v)
|
|
}
|
|
if len(rest) == 0 {
|
|
return name
|
|
}
|
|
sort.Strings(rest)
|
|
return name + " (" + strings.Join(rest, ", ") + ")"
|
|
}
|