Files
terdut-server/internal/api/incident_store.go
T
Niklas Ye f1754c583e Flag incidents that carry notes in the queue
A blue tag with the notebook icon and a count marks an incident with
working notes; a green one marks a note on what fixed it. Both let you
scan the queue for incidents that have more to say than their title,
without opening each one.

The incident list and detail JSON gain note_count and
resolution_note_count, counted in the same query that selects the
incident, so a list costs no extra request per row. The fields are
additive: terdut-tui and terdut-operator ignore them and need no change.

Claude-Session: https://claude.ai/code/session_016mBLURvJoMuUEr9cB2RpUN
2026-10-10 13:29:08 +02:00

408 lines
15 KiB
Go

package api
import (
"context"
"database/sql"
"encoding/json"
"sort"
"strings"
"time"
"git.ryuvia.com/niklas/terdut-server/internal/models"
)
// Values for incidents.resolution_source, recording who closed the incident:
// every member alert stopped firing, or a person decided it was done.
const (
incidentResolutionAlerts = "alerts"
incidentResolutionManual = "manual"
// incidentResolutionRecovered closes a dead man's switch incident whose
// heartbeat started arriving again. It cannot be "alerts": these incidents
// have no member alerts for the cascade to work from.
incidentResolutionRecovered = "recovered"
)
// Incident timeline event types. Stored as free text so adding one later is not
// a migration, but these are the ones the server writes.
const (
evTriggered = "triggered"
evAlertAdded = "alert_added"
evAlertResolved = "alert_resolved"
evAcknowledged = "acknowledged"
evUnacknowledged = "unacknowledged"
evAssigned = "assigned"
evArchived = "archived"
evUnarchived = "unarchived"
evSnoozed = "snoozed"
evUnsnoozed = "unsnoozed"
evResolved = "resolved"
evNote = "note"
// evResolutionNote is the note worth finding again: what fixed it. The
// similar-incidents lookup and the page lead with these; plain notes are
// the working chatter and stay one click away.
evResolutionNote = "resolution_note"
evDeadmanSilent = "deadman_silent"
)
// severityLabel is the Alertmanager label an incident's severity is derived from.
const severityLabel = "severity"
// querier is satisfied by both *sql.DB and *sql.Tx, so the helpers below work
// inside the webhook's transaction and standalone from handlers and the sweeper.
type querier interface {
ExecContext(ctx context.Context, query string, args ...any) (sql.Result, error)
QueryContext(ctx context.Context, query string, args ...any) (*sql.Rows, error)
QueryRowContext(ctx context.Context, query string, args ...any) *sql.Row
}
const incidentSelectFrom = `
SELECT i.id, i.team_id, t.name, i.group_key, i.title, i.group_labels, i.status, i.severity,
i.escalation_level,
-- When this level runs out. Computed here rather than in Go because
-- the timeout lives beside the level in the policy, and one join is
-- cheaper than a second query per incident in a list.
(SELECT i.escalation_level_at + el.timeout_seconds
FROM escalation_levels el
WHERE el.team_id = i.team_id AND el.position = i.escalation_level),
i.triggered_at,
i.acknowledged_by, i.acknowledged_at, ack.username,
i.acknowledged_by_service_account_id, acksa.name,
i.assigned_to, asg.username, i.snoozed_until,
i.resolved_at, i.resolution_source, i.archived_at,
-- How many notes, so a list can flag the incidents that carry extra
-- information without fetching each timeline.
(SELECT count(*) FROM incident_events e WHERE e.incident_id = i.id AND e.type = 'note'),
(SELECT count(*) FROM incident_events e WHERE e.incident_id = i.id AND e.type = 'resolution_note')
FROM incidents i
JOIN teams t ON t.id = i.team_id
LEFT JOIN users ack ON ack.id = i.acknowledged_by
LEFT JOIN service_accounts acksa ON acksa.id = i.acknowledged_by_service_account_id
LEFT JOIN users asg ON asg.id = i.assigned_to`
func scanIncident(s scanner) (models.Incident, error) {
var i models.Incident
var groupLabelsJSON string
var triggeredAt int64
var ackAt, snoozedUntil, resolvedAt, archivedAt, escalationDue *int64
if err := s.Scan(
&i.ID, &i.TeamID, &i.TeamName, &i.GroupKey, &i.Title, &groupLabelsJSON, &i.Status, &i.Severity,
&i.EscalationLevel, &escalationDue,
&triggeredAt,
&i.AcknowledgedByID, &ackAt, &i.AcknowledgedByUser,
&i.AcknowledgedByServiceAccountID, &i.AcknowledgedByServiceAccountName,
&i.AssignedToID, &i.AssignedToUser, &snoozedUntil,
&resolvedAt, &i.ResolutionSource, &archivedAt,
&i.NoteCount, &i.ResolutionNoteCount,
); err != nil {
return i, err
}
json.Unmarshal([]byte(groupLabelsJSON), &i.GroupLabels) //nolint:errcheck
i.TriggeredAt = time.Unix(triggeredAt, 0).UTC()
i.AcknowledgedAt = unixPtr(ackAt)
i.SnoozedUntil = unixPtr(snoozedUntil)
i.ResolvedAt = unixPtr(resolvedAt)
i.ArchivedAt = unixPtr(archivedAt)
i.EscalationDueAt = unixPtr(escalationDue)
return i, nil
}
// unixPtr converts a nullable Unix-second column to a nullable UTC time.
func unixPtr(sec *int64) *time.Time {
if sec == nil {
return nil
}
t := time.Unix(*sec, 0).UTC()
return &t
}
func fetchIncident(ctx context.Context, q querier, id int64) (models.Incident, error) {
return scanIncident(q.QueryRowContext(ctx, incidentSelectFrom+" WHERE i.id = $1", id))
}
// callerActorIDs resolves the current request's caller into the pair of
// nilable ids logEvent/acknowledgeIncidentAs expect: exactly one of userID/
// serviceAccountID is set (never both), replacing the unchecked
// userFromContext(ctx) zero-value reads that used to write a human-only id
// of 0 for a service-account caller (terdut-server#25).
func callerActorIDs(ctx context.Context) (userID, serviceAccountID *int64) {
caller, _ := callerFromContext(ctx)
if u, ok := caller.AsHuman(); ok {
return &u.ID, nil
}
if id, ok := caller.ServiceAccountID(); ok {
return nil, &id
}
return nil, nil
}
// logEvent appends one entry to an incident's timeline. userID and
// serviceAccountID are mutually exclusive and both nilable; both nil means
// the server acted rather than any caller (see incident_events_actor_xor_chk,
// migration 015).
func logEvent(ctx context.Context, q querier, incidentID int64, evType string, userID, serviceAccountID, alertID *int64, detail *string) error {
_, err := q.ExecContext(ctx, `
INSERT INTO incident_events (incident_id, type, user_id, service_account_id, alert_id, detail, created_at)
VALUES ($1, $2, $3, $4, $5, $6, $7)`,
incidentID, evType, userID, serviceAccountID, alertID, detail, time.Now().Unix())
return err
}
// logAssignedEvent records an assignment: user_id is the assignee, and the
// caller who performed it goes in the actor_* columns (migration 018), since
// user_id cannot hold both.
func logAssignedEvent(ctx context.Context, q querier, incidentID, assigneeID int64, actorUserID, actorServiceAccountID *int64) error {
_, err := q.ExecContext(ctx, `
INSERT INTO incident_events (incident_id, type, user_id, actor_user_id, actor_service_account_id, created_at)
VALUES ($1, $2, $3, $4, $5, $6)`,
incidentID, evAssigned, assigneeID, actorUserID, actorServiceAccountID, time.Now().Unix())
return err
}
// todayUTC is the schedule's day key. The schedule's smallest unit is one UTC day.
func todayUTC() string {
return time.Now().UTC().Format("2006-01-02")
}
// currentOnCall returns a team's on-call user for today, or nil when nobody is
// scheduled. A missing schedule entry is not an error — incidents just open
// unassigned.
//
// Per team: each team keeps its own rota, so two teams can have two different
// people on call on the same day, which was the point of scoping the schedule.
func currentOnCall(ctx context.Context, q querier, teamID int64) (*int64, error) {
var userID int64
err := q.QueryRowContext(ctx,
"SELECT user_id FROM schedule_entries WHERE team_id = $1 AND date = $2",
teamID, todayUTC()).Scan(&userID)
if err == sql.ErrNoRows {
return nil, nil
}
if err != nil {
return nil, err
}
return &userID, nil
}
// severityRank orders the conventional Alertmanager severity label values.
// Anything unrecognised sorts below all of them rather than being dropped.
func severityRank(s string) int {
switch strings.ToLower(s) {
case "critical":
return 4
case "error":
return 3
case "warning":
return 2
case "info":
return 1
default:
return 0
}
}
// refreshSeverity raises an incident's severity to the highest `severity` label
// seen across its alerts.
//
// It is a high-water mark, never lowered: an incident that hit critical was a
// critical incident, even after the critical alert clears and a warning is all
// that is left firing. Downgrading a live incident would also quietly demote it
// in the queue while the work is still open.
func refreshSeverity(ctx context.Context, q querier, incidentID int64) error {
rows, err := q.QueryContext(ctx, `
SELECT a.labels ->> $1
FROM incident_alerts ia
JOIN alerts a ON a.id = ia.alert_id
WHERE ia.incident_id = $2`, severityLabel, incidentID)
if err != nil {
return err
}
best := ""
for rows.Next() {
var sev *string
if err := rows.Scan(&sev); err != nil {
rows.Close()
return err
}
if sev != nil && severityRank(*sev) > severityRank(best) {
best = *sev
}
}
if err := rows.Err(); err != nil {
rows.Close()
return err
}
rows.Close()
if best == "" {
return nil
}
// The comparison lives in SQL so an unrelated concurrent update cannot be
// clobbered by a stale read.
_, err = q.ExecContext(ctx, `
UPDATE incidents SET severity = $1
WHERE id = $2
AND (severity IS NULL OR `+severityRankSQL("severity")+` < $3)`,
best, incidentID, severityRank(best))
return err
}
// severityRankSQL mirrors severityRank for use inside a statement. SQL cannot
// order these strings meaningfully on its own.
func severityRankSQL(col string) string {
return `CASE lower(COALESCE(` + col + `, ''))
WHEN 'critical' THEN 4
WHEN 'error' THEN 3
WHEN 'warning' THEN 2
WHEN 'info' THEN 1
ELSE 0 END`
}
// resolveIfSettled closes an incident once every alert under it has stopped
// firing — PagerDuty's cascade, and the only automatic route out of the open
// state. Reports whether it actually resolved anything.
func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, error) {
res, err := q.ExecContext(ctx, `
UPDATE incidents
SET status = 'resolved',
resolved_at = $1,
resolution_source = $2
WHERE id = $3
AND resolved_at IS NULL
-- An incident with no members yet is mid-creation, not settled.
AND EXISTS (SELECT 1 FROM incident_alerts ia WHERE ia.incident_id = incidents.id)
AND NOT EXISTS (SELECT 1
FROM incident_alerts ia
JOIN alerts a ON a.id = ia.alert_id
WHERE ia.incident_id = incidents.id
AND a.status = 'firing')`,
time.Now().Unix(), incidentResolutionAlerts, incidentID)
if err != nil {
return false, err
}
n, _ := res.RowsAffected()
if n == 0 {
return false, nil
}
if err := stopEscalation(ctx, q, incidentID); err != nil {
return false, err
}
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil, nil); err != nil {
return false, err
}
// The all-clear goes only to whoever was paged in the first place, which
// enqueueResolved works out from the incident's own notification history.
// Manual resolution sends nothing: the person who closed it already knows.
return true, enqueueResolved(ctx, q, incidentID)
}
// acknowledgeIncident records that userID — a human — has picked an incident
// up, and reports whether it changed anything — an already-resolved or
// already-acknowledged incident is left alone, so a second acknowledge (a
// retried request, or a stale push notification tapped after the web UI
// already acked it) is a no-op rather than a second "acknowledged" timeline
// entry. Used only by the Acknowledge button in a push notification
// (notify_ack.go), which always resolves a human from
// incident_ack_tokens.user_id — there is no service-account equivalent of
// that flow, so this keeps its human-only signature; the authenticated
// handler goes through acknowledgeIncidentAs below instead.
func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int64) (bool, error) {
return acknowledgeIncidentAs(ctx, q, incidentID, &userID, nil)
}
// acknowledgeIncidentAs is acknowledgeIncident generalized to either actor
// kind. userID and serviceAccountID are mutually exclusive and nilable the
// same way logEvent's are (see incidents_ack_actor_xor_chk, migration 015).
func acknowledgeIncidentAs(ctx context.Context, q querier, incidentID int64, userID, serviceAccountID *int64) (bool, error) {
res, err := q.ExecContext(ctx, `
UPDATE incidents
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_by_service_account_id = $2,
acknowledged_at = $3
WHERE id = $4 AND status = 'triggered'`,
userID, serviceAccountID, time.Now().Unix(), incidentID)
if err != nil {
return false, err
}
if n, _ := res.RowsAffected(); n == 0 {
return false, nil
}
// Somebody has it: stop waking anybody else.
if err := stopEscalation(ctx, q, incidentID); err != nil {
return false, err
}
return true, logEvent(ctx, q, incidentID, evAcknowledged, userID, serviceAccountID, nil, nil)
}
// openIncidentForAlert returns the open incident an alert currently belongs to,
// or 0 when it has none. Used when an alert resolves or expires so the event
// lands on the right timeline.
func openIncidentForAlert(ctx context.Context, q querier, alertID int64) (int64, error) {
var id int64
err := q.QueryRowContext(ctx, `
SELECT i.id
FROM incident_alerts ia
JOIN incidents i ON i.id = ia.incident_id
WHERE ia.alert_id = $1 AND i.resolved_at IS NULL`, alertID).Scan(&id)
if err == sql.ErrNoRows {
return 0, nil
}
return id, err
}
// volatileLabels say where a problem ran this time, not what the problem is, so
// they stay out of the signature. Migration 008's backfill lists the same set.
var volatileLabels = map[string]bool{
"instance": true, "pod": true, "pod_name": true, "pod_ip": true,
"container": true, "container_name": true, "endpoint": true,
}
// incidentSignature identifies "the same problem" across incidents: the alert
// name plus the stable group labels, sorted. Incidents in one team with equal
// signatures are what the similar-incidents lookup returns. title stands in for
// the name when the payload carried no alertname (groupless and dead man's
// switch incidents).
func incidentSignature(groupLabels map[string]string, title string) string {
name := groupLabels["alertname"]
if name == "" {
name = title
}
rest := make([]string, 0, len(groupLabels))
for k, v := range groupLabels {
if k == "alertname" || volatileLabels[k] {
continue
}
rest = append(rest, k+"="+v)
}
sort.Strings(rest)
return name + "|" + strings.Join(rest, ",")
}
// incidentTitle renders a human-readable title from Alertmanager's groupLabels,
// leading with the alert name and appending whatever else the operator grouped
// by. Falls back to the alert's own name when the payload carried no groupLabels.
func incidentTitle(groupLabels map[string]string, fallback string) string {
name := groupLabels["alertname"]
if name == "" {
name = fallback
}
if name == "" {
name = "Incident"
}
rest := make([]string, 0, len(groupLabels))
for k, v := range groupLabels {
if k == "alertname" {
continue
}
rest = append(rest, k+"="+v)
}
if len(rest) == 0 {
return name
}
sort.Strings(rest)
return name + " (" + strings.Join(rest, ", ") + ")"
}