9da913080f
Every incident-mutation handler read userFromContext(ctx) and wrote the result's .ID into acknowledged_by/incident_events.user_id without checking the ok bool. For a team-scoped service-account caller this returned a zero-value user id, which violated the users(id) FK and 500'd on acknowledge, unacknowledge, resolve, snooze, unsnooze and create-note. handleDeleteNote didn't crash but silently matched zero rows instead (WHERE user_id = 0), so a service account could never delete its own note. Add acknowledged_by_service_account_id (incidents) and service_account_id (incident_events) as nullable FKs to service_accounts(id), parallel to and mutually exclusive with the existing human columns (migration 015, with a CHECK enforcing the exclusion). Route every one of the six handlers plus delete-note through a new callerActorIDs() helper that branches on Caller.AsHuman()/ServiceAccountID() instead of assuming a human, and thread a serviceAccountID parameter through logEvent and the new acknowledgeIncidentAs (acknowledgeIncident itself is untouched: its only other caller, the push-notification Acknowledge button, is always human). Render the new actor distinctly from both a human and "the server acted" in the web UI's incident timeline and facts card. handleIncidentAssign, handleIncidentArchive and handleIncidentUnarchive are deliberately not touched here — they track no actor at all today, for anyone, which is a separate pre-existing gap (follow-up issue to come). Fixes #25
390 lines
14 KiB
Go
390 lines
14 KiB
Go
package api
|
|
|
|
import (
|
|
"context"
|
|
"database/sql"
|
|
"encoding/json"
|
|
"sort"
|
|
"strings"
|
|
"time"
|
|
|
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
|
)
|
|
|
|
// Values for incidents.resolution_source, recording who closed the incident:
|
|
// every member alert stopped firing, or a person decided it was done.
|
|
const (
|
|
incidentResolutionAlerts = "alerts"
|
|
incidentResolutionManual = "manual"
|
|
|
|
// incidentResolutionRecovered closes a dead man's switch incident whose
|
|
// heartbeat started arriving again. It cannot be "alerts": these incidents
|
|
// have no member alerts for the cascade to work from.
|
|
incidentResolutionRecovered = "recovered"
|
|
)
|
|
|
|
// Incident timeline event types. Stored as free text so adding one later is not
|
|
// a migration, but these are the ones the server writes.
|
|
const (
|
|
evTriggered = "triggered"
|
|
evAlertAdded = "alert_added"
|
|
evAlertResolved = "alert_resolved"
|
|
evAcknowledged = "acknowledged"
|
|
evUnacknowledged = "unacknowledged"
|
|
evAssigned = "assigned"
|
|
evSnoozed = "snoozed"
|
|
evUnsnoozed = "unsnoozed"
|
|
evResolved = "resolved"
|
|
evNote = "note"
|
|
// evResolutionNote is the note worth finding again: what fixed it. The
|
|
// similar-incidents lookup and the page lead with these; plain notes are
|
|
// the working chatter and stay one click away.
|
|
evResolutionNote = "resolution_note"
|
|
evDeadmanSilent = "deadman_silent"
|
|
)
|
|
|
|
// severityLabel is the Alertmanager label an incident's severity is derived from.
|
|
const severityLabel = "severity"
|
|
|
|
// querier is satisfied by both *sql.DB and *sql.Tx, so the helpers below work
|
|
// inside the webhook's transaction and standalone from handlers and the sweeper.
|
|
type querier interface {
|
|
ExecContext(ctx context.Context, query string, args ...any) (sql.Result, error)
|
|
QueryContext(ctx context.Context, query string, args ...any) (*sql.Rows, error)
|
|
QueryRowContext(ctx context.Context, query string, args ...any) *sql.Row
|
|
}
|
|
|
|
const incidentSelectFrom = `
|
|
SELECT i.id, i.team_id, t.name, i.group_key, i.title, i.group_labels, i.status, i.severity,
|
|
i.escalation_level,
|
|
-- When this level runs out. Computed here rather than in Go because
|
|
-- the timeout lives beside the level in the policy, and one join is
|
|
-- cheaper than a second query per incident in a list.
|
|
(SELECT i.escalation_level_at + el.timeout_seconds
|
|
FROM escalation_levels el
|
|
WHERE el.team_id = i.team_id AND el.position = i.escalation_level),
|
|
i.triggered_at,
|
|
i.acknowledged_by, i.acknowledged_at, ack.username,
|
|
i.acknowledged_by_service_account_id, acksa.name,
|
|
i.assigned_to, asg.username, i.snoozed_until,
|
|
i.resolved_at, i.resolution_source, i.archived_at
|
|
FROM incidents i
|
|
JOIN teams t ON t.id = i.team_id
|
|
LEFT JOIN users ack ON ack.id = i.acknowledged_by
|
|
LEFT JOIN service_accounts acksa ON acksa.id = i.acknowledged_by_service_account_id
|
|
LEFT JOIN users asg ON asg.id = i.assigned_to`
|
|
|
|
func scanIncident(s scanner) (models.Incident, error) {
|
|
var i models.Incident
|
|
var groupLabelsJSON string
|
|
var triggeredAt int64
|
|
var ackAt, snoozedUntil, resolvedAt, archivedAt, escalationDue *int64
|
|
|
|
if err := s.Scan(
|
|
&i.ID, &i.TeamID, &i.TeamName, &i.GroupKey, &i.Title, &groupLabelsJSON, &i.Status, &i.Severity,
|
|
&i.EscalationLevel, &escalationDue,
|
|
&triggeredAt,
|
|
&i.AcknowledgedByID, &ackAt, &i.AcknowledgedByUser,
|
|
&i.AcknowledgedByServiceAccountID, &i.AcknowledgedByServiceAccountName,
|
|
&i.AssignedToID, &i.AssignedToUser, &snoozedUntil,
|
|
&resolvedAt, &i.ResolutionSource, &archivedAt,
|
|
); err != nil {
|
|
return i, err
|
|
}
|
|
|
|
json.Unmarshal([]byte(groupLabelsJSON), &i.GroupLabels) //nolint:errcheck
|
|
i.TriggeredAt = time.Unix(triggeredAt, 0).UTC()
|
|
i.AcknowledgedAt = unixPtr(ackAt)
|
|
i.SnoozedUntil = unixPtr(snoozedUntil)
|
|
i.ResolvedAt = unixPtr(resolvedAt)
|
|
i.ArchivedAt = unixPtr(archivedAt)
|
|
i.EscalationDueAt = unixPtr(escalationDue)
|
|
return i, nil
|
|
}
|
|
|
|
// unixPtr converts a nullable Unix-second column to a nullable UTC time.
|
|
func unixPtr(sec *int64) *time.Time {
|
|
if sec == nil {
|
|
return nil
|
|
}
|
|
t := time.Unix(*sec, 0).UTC()
|
|
return &t
|
|
}
|
|
|
|
func fetchIncident(ctx context.Context, q querier, id int64) (models.Incident, error) {
|
|
return scanIncident(q.QueryRowContext(ctx, incidentSelectFrom+" WHERE i.id = $1", id))
|
|
}
|
|
|
|
// callerActorIDs resolves the current request's caller into the pair of
|
|
// nilable ids logEvent/acknowledgeIncidentAs expect: exactly one of userID/
|
|
// serviceAccountID is set (never both), replacing the unchecked
|
|
// userFromContext(ctx) zero-value reads that used to write a human-only id
|
|
// of 0 for a service-account caller (terdut-server#25).
|
|
func callerActorIDs(ctx context.Context) (userID, serviceAccountID *int64) {
|
|
caller, _ := callerFromContext(ctx)
|
|
if u, ok := caller.AsHuman(); ok {
|
|
return &u.ID, nil
|
|
}
|
|
if id, ok := caller.ServiceAccountID(); ok {
|
|
return nil, &id
|
|
}
|
|
return nil, nil
|
|
}
|
|
|
|
// logEvent appends one entry to an incident's timeline. userID and
|
|
// serviceAccountID are mutually exclusive and both nilable; both nil means
|
|
// the server acted rather than any caller (see incident_events_actor_xor_chk,
|
|
// migration 015).
|
|
func logEvent(ctx context.Context, q querier, incidentID int64, evType string, userID, serviceAccountID, alertID *int64, detail *string) error {
|
|
_, err := q.ExecContext(ctx, `
|
|
INSERT INTO incident_events (incident_id, type, user_id, service_account_id, alert_id, detail, created_at)
|
|
VALUES ($1, $2, $3, $4, $5, $6, $7)`,
|
|
incidentID, evType, userID, serviceAccountID, alertID, detail, time.Now().Unix())
|
|
return err
|
|
}
|
|
|
|
// todayUTC is the schedule's day key. The schedule's smallest unit is one UTC day.
|
|
func todayUTC() string {
|
|
return time.Now().UTC().Format("2006-01-02")
|
|
}
|
|
|
|
// currentOnCall returns a team's on-call user for today, or nil when nobody is
|
|
// scheduled. A missing schedule entry is not an error — incidents just open
|
|
// unassigned.
|
|
//
|
|
// Per team: each team keeps its own rota, so two teams can have two different
|
|
// people on call on the same day, which was the point of scoping the schedule.
|
|
func currentOnCall(ctx context.Context, q querier, teamID int64) (*int64, error) {
|
|
var userID int64
|
|
err := q.QueryRowContext(ctx,
|
|
"SELECT user_id FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
|
teamID, todayUTC()).Scan(&userID)
|
|
if err == sql.ErrNoRows {
|
|
return nil, nil
|
|
}
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return &userID, nil
|
|
}
|
|
|
|
// severityRank orders the conventional Alertmanager severity label values.
|
|
// Anything unrecognised sorts below all of them rather than being dropped.
|
|
func severityRank(s string) int {
|
|
switch strings.ToLower(s) {
|
|
case "critical":
|
|
return 4
|
|
case "error":
|
|
return 3
|
|
case "warning":
|
|
return 2
|
|
case "info":
|
|
return 1
|
|
default:
|
|
return 0
|
|
}
|
|
}
|
|
|
|
// refreshSeverity raises an incident's severity to the highest `severity` label
|
|
// seen across its alerts.
|
|
//
|
|
// It is a high-water mark, never lowered: an incident that hit critical was a
|
|
// critical incident, even after the critical alert clears and a warning is all
|
|
// that is left firing. Downgrading a live incident would also quietly demote it
|
|
// in the queue while the work is still open.
|
|
func refreshSeverity(ctx context.Context, q querier, incidentID int64) error {
|
|
rows, err := q.QueryContext(ctx, `
|
|
SELECT a.labels ->> $1
|
|
FROM incident_alerts ia
|
|
JOIN alerts a ON a.id = ia.alert_id
|
|
WHERE ia.incident_id = $2`, severityLabel, incidentID)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
best := ""
|
|
for rows.Next() {
|
|
var sev *string
|
|
if err := rows.Scan(&sev); err != nil {
|
|
rows.Close()
|
|
return err
|
|
}
|
|
if sev != nil && severityRank(*sev) > severityRank(best) {
|
|
best = *sev
|
|
}
|
|
}
|
|
if err := rows.Err(); err != nil {
|
|
rows.Close()
|
|
return err
|
|
}
|
|
rows.Close()
|
|
|
|
if best == "" {
|
|
return nil
|
|
}
|
|
// The comparison lives in SQL so an unrelated concurrent update cannot be
|
|
// clobbered by a stale read.
|
|
_, err = q.ExecContext(ctx, `
|
|
UPDATE incidents SET severity = $1
|
|
WHERE id = $2
|
|
AND (severity IS NULL OR `+severityRankSQL("severity")+` < $3)`,
|
|
best, incidentID, severityRank(best))
|
|
return err
|
|
}
|
|
|
|
// severityRankSQL mirrors severityRank for use inside a statement. SQL cannot
|
|
// order these strings meaningfully on its own.
|
|
func severityRankSQL(col string) string {
|
|
return `CASE lower(COALESCE(` + col + `, ''))
|
|
WHEN 'critical' THEN 4
|
|
WHEN 'error' THEN 3
|
|
WHEN 'warning' THEN 2
|
|
WHEN 'info' THEN 1
|
|
ELSE 0 END`
|
|
}
|
|
|
|
// resolveIfSettled closes an incident once every alert under it has stopped
|
|
// firing — PagerDuty's cascade, and the only automatic route out of the open
|
|
// state. Reports whether it actually resolved anything.
|
|
func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, error) {
|
|
res, err := q.ExecContext(ctx, `
|
|
UPDATE incidents
|
|
SET status = 'resolved',
|
|
resolved_at = $1,
|
|
resolution_source = $2
|
|
WHERE id = $3
|
|
AND resolved_at IS NULL
|
|
-- An incident with no members yet is mid-creation, not settled.
|
|
AND EXISTS (SELECT 1 FROM incident_alerts ia WHERE ia.incident_id = incidents.id)
|
|
AND NOT EXISTS (SELECT 1
|
|
FROM incident_alerts ia
|
|
JOIN alerts a ON a.id = ia.alert_id
|
|
WHERE ia.incident_id = incidents.id
|
|
AND a.status = 'firing')`,
|
|
time.Now().Unix(), incidentResolutionAlerts, incidentID)
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
n, _ := res.RowsAffected()
|
|
if n == 0 {
|
|
return false, nil
|
|
}
|
|
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
|
return false, err
|
|
}
|
|
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil, nil); err != nil {
|
|
return false, err
|
|
}
|
|
// The all-clear goes only to whoever was paged in the first place, which
|
|
// enqueueResolved works out from the incident's own notification history.
|
|
// Manual resolution sends nothing: the person who closed it already knows.
|
|
return true, enqueueResolved(ctx, q, incidentID)
|
|
}
|
|
|
|
// acknowledgeIncident records that userID — a human — has picked an incident
|
|
// up, and reports whether it changed anything — an already-resolved or
|
|
// already-acknowledged incident is left alone, so a second acknowledge (a
|
|
// retried request, or a stale push notification tapped after the web UI
|
|
// already acked it) is a no-op rather than a second "acknowledged" timeline
|
|
// entry. Used only by the Acknowledge button in a push notification
|
|
// (notify_ack.go), which always resolves a human from
|
|
// incident_ack_tokens.user_id — there is no service-account equivalent of
|
|
// that flow, so this keeps its human-only signature; the authenticated
|
|
// handler goes through acknowledgeIncidentAs below instead.
|
|
func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int64) (bool, error) {
|
|
return acknowledgeIncidentAs(ctx, q, incidentID, &userID, nil)
|
|
}
|
|
|
|
// acknowledgeIncidentAs is acknowledgeIncident generalized to either actor
|
|
// kind. userID and serviceAccountID are mutually exclusive and nilable the
|
|
// same way logEvent's are (see incidents_ack_actor_xor_chk, migration 015).
|
|
func acknowledgeIncidentAs(ctx context.Context, q querier, incidentID int64, userID, serviceAccountID *int64) (bool, error) {
|
|
res, err := q.ExecContext(ctx, `
|
|
UPDATE incidents
|
|
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_by_service_account_id = $2,
|
|
acknowledged_at = $3
|
|
WHERE id = $4 AND status = 'triggered'`,
|
|
userID, serviceAccountID, time.Now().Unix(), incidentID)
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
if n, _ := res.RowsAffected(); n == 0 {
|
|
return false, nil
|
|
}
|
|
// Somebody has it: stop waking anybody else.
|
|
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
|
return false, err
|
|
}
|
|
return true, logEvent(ctx, q, incidentID, evAcknowledged, userID, serviceAccountID, nil, nil)
|
|
}
|
|
|
|
// openIncidentForAlert returns the open incident an alert currently belongs to,
|
|
// or 0 when it has none. Used when an alert resolves or expires so the event
|
|
// lands on the right timeline.
|
|
func openIncidentForAlert(ctx context.Context, q querier, alertID int64) (int64, error) {
|
|
var id int64
|
|
err := q.QueryRowContext(ctx, `
|
|
SELECT i.id
|
|
FROM incident_alerts ia
|
|
JOIN incidents i ON i.id = ia.incident_id
|
|
WHERE ia.alert_id = $1 AND i.resolved_at IS NULL`, alertID).Scan(&id)
|
|
if err == sql.ErrNoRows {
|
|
return 0, nil
|
|
}
|
|
return id, err
|
|
}
|
|
|
|
// volatileLabels say where a problem ran this time, not what the problem is, so
|
|
// they stay out of the signature. Migration 008's backfill lists the same set.
|
|
var volatileLabels = map[string]bool{
|
|
"instance": true, "pod": true, "pod_name": true, "pod_ip": true,
|
|
"container": true, "container_name": true, "endpoint": true,
|
|
}
|
|
|
|
// incidentSignature identifies "the same problem" across incidents: the alert
|
|
// name plus the stable group labels, sorted. Incidents in one team with equal
|
|
// signatures are what the similar-incidents lookup returns. title stands in for
|
|
// the name when the payload carried no alertname (groupless and dead man's
|
|
// switch incidents).
|
|
func incidentSignature(groupLabels map[string]string, title string) string {
|
|
name := groupLabels["alertname"]
|
|
if name == "" {
|
|
name = title
|
|
}
|
|
rest := make([]string, 0, len(groupLabels))
|
|
for k, v := range groupLabels {
|
|
if k == "alertname" || volatileLabels[k] {
|
|
continue
|
|
}
|
|
rest = append(rest, k+"="+v)
|
|
}
|
|
sort.Strings(rest)
|
|
return name + "|" + strings.Join(rest, ",")
|
|
}
|
|
|
|
// incidentTitle renders a human-readable title from Alertmanager's groupLabels,
|
|
// leading with the alert name and appending whatever else the operator grouped
|
|
// by. Falls back to the alert's own name when the payload carried no groupLabels.
|
|
func incidentTitle(groupLabels map[string]string, fallback string) string {
|
|
name := groupLabels["alertname"]
|
|
if name == "" {
|
|
name = fallback
|
|
}
|
|
if name == "" {
|
|
name = "Incident"
|
|
}
|
|
|
|
rest := make([]string, 0, len(groupLabels))
|
|
for k, v := range groupLabels {
|
|
if k == "alertname" {
|
|
continue
|
|
}
|
|
rest = append(rest, k+"="+v)
|
|
}
|
|
if len(rest) == 0 {
|
|
return name
|
|
}
|
|
sort.Strings(rest)
|
|
return name + " (" + strings.Join(rest, ", ") + ")"
|
|
}
|