Compare commits
14 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| df83adfe47 | |||
| 9da913080f | |||
| 2b2609e98f | |||
| fa6d82d6e5 | |||
| 7efd1bbba7 | |||
| 3bf94a5d7f | |||
| 5c4e0bdd0e | |||
| 9e5b085d8b | |||
| 0050738ca0 | |||
| 42180948d1 | |||
| c83c7c2a8b | |||
| 43beda9a30 | |||
| 710521a73c | |||
| d9492913ed |
@@ -15,5 +15,5 @@ type: application
|
||||
# appVersion and image.tag in values.yaml no longer agree, and that is not an oversight:
|
||||
# image.tag stays "latest", which is what a local install actually pulls. appVersion is
|
||||
# metadata and drives nothing.
|
||||
version: 0.35.0
|
||||
appVersion: "v0.35.0"
|
||||
version: 0.37.2
|
||||
appVersion: "v0.37.2"
|
||||
|
||||
@@ -6,15 +6,19 @@ metadata:
|
||||
labels:
|
||||
{{- include "terdut-server.labels" . | nindent 4 }}
|
||||
spec:
|
||||
replicas: 1
|
||||
replicas: {{ .Values.replicaCount }}
|
||||
selector:
|
||||
matchLabels:
|
||||
{{- include "terdut-server.selectorLabels" . | nindent 6 }}
|
||||
# Recreate, not RollingUpdate, even though the PVC that forced it is gone: the
|
||||
# sweeper and the notifier are unsynchronised singletons, and two replicas
|
||||
# overlapping during a rollout would both page for the same incident.
|
||||
# RollingUpdate, not Recreate: the sweeper, notifier and migration runner
|
||||
# each take a Postgres advisory lock around their own pass, and new-incident
|
||||
# creation on the first webhook for a brand-new groupKey resolves its own
|
||||
# insert conflict -- so two replicas overlapping during a rollout no longer
|
||||
# double-page, race a migration, or drop a webhook payload (v0.36.0). No
|
||||
# explicit maxUnavailable/maxSurge: the 25%/25% default rounds to 0/1 at
|
||||
# replicaCount: 2, which is zero-downtime already.
|
||||
strategy:
|
||||
type: Recreate
|
||||
type: RollingUpdate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
|
||||
@@ -1,3 +1,10 @@
|
||||
# Safe above 1 since v0.36.0: the sweeper, notifier and migration runner each
|
||||
# take a Postgres advisory lock around their own pass, and a webhook that
|
||||
# loses the race to open a brand-new incident attaches to the winner's row
|
||||
# instead of dropping its payload. An image older than v0.36.0 does not have
|
||||
# these guards -- do not raise this against one.
|
||||
replicaCount: 2
|
||||
|
||||
networking:
|
||||
hostname: "terdut.example.com"
|
||||
servicePort: 8080
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
package api
|
||||
|
||||
// This file is internal (package api, not api_test) because withAdvisoryLock is
|
||||
// unexported and these tests exercise its locking semantics directly rather than
|
||||
// through the full StartArchiver/StartNotifier loop, which would make the "does
|
||||
// not run while held" case timing-dependent instead of deterministic. It opens a
|
||||
// plain connection to TERDUT_TEST_DSN rather than reusing testdb_test.go's
|
||||
// newTestDB, since that helper lives in the separate, already-compiled
|
||||
// api_test package and a Postgres advisory lock needs no schema or migration
|
||||
// to exercise.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
_ "github.com/jackc/pgx/v5/stdlib"
|
||||
)
|
||||
|
||||
// advisoryTestDB opens a plain, unmigrated connection to the test database. An
|
||||
// unset DSN fails rather than skips, matching testdb_test.go's rationale: a
|
||||
// suite that quietly tests nothing is worse than one that does not run.
|
||||
func advisoryTestDB(t *testing.T) *sql.DB {
|
||||
t.Helper()
|
||||
|
||||
dsn := os.Getenv("TERDUT_TEST_DSN")
|
||||
if dsn == "" {
|
||||
t.Fatalf("TERDUT_TEST_DSN is not set: these tests need Postgres.\n" +
|
||||
"Run `make test-db` for a local one, then\n" +
|
||||
" export TERDUT_TEST_DSN=postgres://terdut:terdut@localhost:5432/terdut_test?sslmode=disable")
|
||||
}
|
||||
|
||||
db, err := sql.Open("pgx", dsn)
|
||||
if err != nil {
|
||||
t.Fatalf("connect to TERDUT_TEST_DSN: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { db.Close() })
|
||||
return db
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_RunsWhenFree(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||
|
||||
if !ran {
|
||||
t.Fatal("fn did not run although the lock was free")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_SkipsWhileHeldElsewhere(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
// Hold the lock on a connection of our own, standing in for another
|
||||
// replica mid-pass.
|
||||
holder, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("acquire holder connection: %v", err)
|
||||
}
|
||||
defer holder.Close()
|
||||
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_lock($1)", archiverLockKey); err != nil {
|
||||
t.Fatalf("pre-acquire lock: %v", err)
|
||||
}
|
||||
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||
if ran {
|
||||
t.Fatal("fn ran although another connection already held the lock")
|
||||
}
|
||||
|
||||
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", archiverLockKey); err != nil {
|
||||
t.Fatalf("release held lock: %v", err)
|
||||
}
|
||||
|
||||
// Now that the holder released it, the next caller should get it.
|
||||
ran = false
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||
if !ran {
|
||||
t.Fatal("fn did not run after the other connection released the lock")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_ReleasesAfterFnReturns(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() {})
|
||||
|
||||
// If the first call had leaked the lock, this one would see it held and
|
||||
// skip, leaving ran false.
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() { ran = true })
|
||||
if !ran {
|
||||
t.Fatal("fn did not run on a later call: the earlier call leaked its lock")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_KeysAreIndependent(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
holder, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("acquire holder connection: %v", err)
|
||||
}
|
||||
defer holder.Close()
|
||||
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_lock($1)", archiverLockKey); err != nil {
|
||||
t.Fatalf("pre-acquire archiver lock: %v", err)
|
||||
}
|
||||
defer holder.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", archiverLockKey)
|
||||
|
||||
// Holding archiverLockKey must not block notifierLockKey.
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() { ran = true })
|
||||
if !ran {
|
||||
t.Fatal("fn did not run under a different key although only archiverLockKey was held")
|
||||
}
|
||||
}
|
||||
@@ -169,7 +169,7 @@ func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, src alertSourc
|
||||
}
|
||||
touched[id] = true
|
||||
alertID := a.id
|
||||
if err := logEvent(ctx, tx, id, evAlertResolved, nil, &alertID, nil); err != nil {
|
||||
if err := logEvent(ctx, tx, id, evAlertResolved, nil, nil, &alertID, nil); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
@@ -376,6 +376,17 @@ func incidentForGroup(ctx context.Context, tx *sql.Tx, notify NotifyConfig, team
|
||||
// its own. Hence the querier rather than a *sql.Tx. A nil severity leaves the
|
||||
// column for refreshSeverity to fill from the member alerts; the sweeper passes
|
||||
// one because its incidents have no members to derive it from.
|
||||
//
|
||||
// Both callers get here only after their own SELECT found no open incident for
|
||||
// this group_key — but on more than one replica, two webhook deliveries for the
|
||||
// very first occurrence of a brand-new group_key can both pass that SELECT
|
||||
// before either INSERTs. ON CONFLICT DO NOTHING against
|
||||
// incidents_open_group_key_idx is what makes the loser's INSERT a no-op instead
|
||||
// of a unique-violation error that would otherwise roll back its entire
|
||||
// payload; existingOpenIncident then hands it the winner's row. Postgres
|
||||
// resolves that conflict only once the winner's transaction has committed (or
|
||||
// rolled back), so by the time this RETURNING comes back empty, the winner's
|
||||
// row is guaranteed visible to that follow-up SELECT.
|
||||
func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID int64, groupKey, title string, groupLabels map[string]string, severity *string) (int64, error) {
|
||||
onCall, err := currentOnCall(ctx, q, teamID)
|
||||
if err != nil {
|
||||
@@ -391,19 +402,27 @@ func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID in
|
||||
err = q.QueryRowContext(ctx, `
|
||||
INSERT INTO incidents (team_id, group_key, title, group_labels, signature, status, severity, triggered_at, assigned_to)
|
||||
VALUES ($1, $2, $3, $4::jsonb, $5, 'triggered', $6, $7, $8)
|
||||
ON CONFLICT (team_id, group_key) WHERE resolved_at IS NULL DO NOTHING
|
||||
RETURNING id`,
|
||||
teamID, groupKey, title, string(labelsJSON), incidentSignature(groupLabels, title), severity,
|
||||
time.Now().Unix(), onCall).Scan(&id)
|
||||
if err != nil {
|
||||
switch {
|
||||
case err == sql.ErrNoRows:
|
||||
// Lost the race: someone else's incident for this group_key exists now.
|
||||
// Everything below — the trigger event, assignment, page, escalation
|
||||
// clock — already happened for that row when it was created; attach to
|
||||
// it rather than fail this call (and the whole payload) outright.
|
||||
return existingOpenIncident(ctx, q, teamID, groupKey)
|
||||
case err != nil:
|
||||
return 0, err
|
||||
}
|
||||
|
||||
if err := logEvent(ctx, q, id, evTriggered, nil, nil, nil); err != nil {
|
||||
if err := logEvent(ctx, q, id, evTriggered, nil, nil, nil, nil); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if onCall != nil {
|
||||
// On an "assigned" event user_id is the assignee, not the actor.
|
||||
if err := logEvent(ctx, q, id, evAssigned, onCall, nil, nil); err != nil {
|
||||
if err := logEvent(ctx, q, id, evAssigned, onCall, nil, nil, nil); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
}
|
||||
@@ -423,6 +442,21 @@ func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID in
|
||||
return id, nil
|
||||
}
|
||||
|
||||
// existingOpenIncident looks up the open incident openIncident's own INSERT just
|
||||
// lost a conflict against — the same lookup incidentForGroup does before ever
|
||||
// calling openIncident, repeated here for the caller that arrived second.
|
||||
func existingOpenIncident(ctx context.Context, q querier, teamID int64, groupKey string) (int64, error) {
|
||||
var id int64
|
||||
err := q.QueryRowContext(ctx,
|
||||
"SELECT id FROM incidents WHERE team_id = $1 AND group_key = $2 AND resolved_at IS NULL",
|
||||
teamID, groupKey,
|
||||
).Scan(&id)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return id, nil
|
||||
}
|
||||
|
||||
// linkAlert adds an alert to an incident, emitting a timeline entry only the
|
||||
// first time. Re-sends of an already-linked alert are silent.
|
||||
func linkAlert(ctx context.Context, tx *sql.Tx, incidentID, alertID int64) error {
|
||||
@@ -436,5 +470,5 @@ func linkAlert(ctx context.Context, tx *sql.Tx, incidentID, alertID int64) error
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
return nil
|
||||
}
|
||||
return logEvent(ctx, tx, incidentID, evAlertAdded, nil, &alertID, nil)
|
||||
return logEvent(ctx, tx, incidentID, evAlertAdded, nil, nil, &alertID, nil)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident reproduces two
|
||||
// replicas racing the very first webhook delivery for a brand-new group_key:
|
||||
// both see no open incident yet (incidentForGroup's own SELECT finds
|
||||
// nothing) and race openIncident's INSERT.
|
||||
//
|
||||
// The DB's own unique index already guarantees at most one incident either
|
||||
// way, with or without this fix — so "exactly one incident" alone cannot
|
||||
// tell a fixed run from a broken one. What ON CONFLICT handling actually
|
||||
// changes is what happens to the *loser*: before it, the loser's INSERT hit
|
||||
// incidents_open_group_key_idx's unique violation, which — since
|
||||
// upsertAlerts ran earlier in that same transaction — rolled back its whole
|
||||
// payload, alert insert included. ingest's error is only logged and
|
||||
// receiveWebhook answers 200 regardless, so nothing ever retried it: the
|
||||
// loser's alert silently never existed. That is the regression signal this
|
||||
// test checks — every caller's fingerprint must show up in /api/alerts, not
|
||||
// just the winner's.
|
||||
func TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
const callers = 8
|
||||
const groupKey = "race-group"
|
||||
|
||||
// Every caller needs its own fingerprint. A shared one would serialize all
|
||||
// of them at upsertAlerts' own ON CONFLICT (team_id, fingerprint) row lock,
|
||||
// long before any of them reached incidentForGroup — which would hide the
|
||||
// very race this test exists to force.
|
||||
bodies := make([][]byte, callers)
|
||||
for i := range callers {
|
||||
payload := map[string]any{
|
||||
"version": "4",
|
||||
"status": "firing",
|
||||
"groupKey": groupKey,
|
||||
"groupLabels": map[string]string{"alertname": "RaceAlert"},
|
||||
"alerts": []map[string]any{amAlert(fmt.Sprintf("fp-race-%d", i), "RaceAlert", "firing",
|
||||
"2026-05-20T10:00:00Z", "0001-01-01T00:00:00Z", nil)},
|
||||
}
|
||||
bodies[i], _ = json.Marshal(payload)
|
||||
}
|
||||
|
||||
// A start line, so every request is fired as close to simultaneously as
|
||||
// goroutine scheduling allows, rather than trickling out one dial at a
|
||||
// time — the race window is the gap between incidentForGroup's SELECT and
|
||||
// openIncident's INSERT, which a staggered start could easily miss.
|
||||
var ready sync.WaitGroup
|
||||
start := make(chan struct{})
|
||||
statuses := make([]int, callers)
|
||||
var wg sync.WaitGroup
|
||||
for i := range callers {
|
||||
ready.Add(1)
|
||||
wg.Add(1)
|
||||
go func(i int) {
|
||||
defer wg.Done()
|
||||
ready.Done()
|
||||
<-start
|
||||
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||
"application/json", bytes.NewReader(bodies[i]))
|
||||
if err != nil {
|
||||
t.Errorf("post webhook #%d: %v", i, err)
|
||||
return
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
statuses[i] = resp.StatusCode
|
||||
}(i)
|
||||
}
|
||||
ready.Wait()
|
||||
close(start)
|
||||
wg.Wait()
|
||||
|
||||
for i, code := range statuses {
|
||||
if code != http.StatusOK {
|
||||
t.Errorf("webhook #%d returned %d, want 200", i, code)
|
||||
}
|
||||
}
|
||||
|
||||
var matched []any
|
||||
for _, inc := range listIncidents(t, s, "") {
|
||||
if inc["group_key"] == groupKey {
|
||||
matched = append(matched, inc["id"])
|
||||
}
|
||||
}
|
||||
if len(matched) != 1 {
|
||||
t.Fatalf("expected exactly 1 incident for group_key %q after %d concurrent deliveries, got %d: %v",
|
||||
groupKey, callers, len(matched), matched)
|
||||
}
|
||||
|
||||
var alerts []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
|
||||
seen := map[string]bool{}
|
||||
for _, a := range alerts {
|
||||
if fp, ok := a["fingerprint"].(string); ok {
|
||||
seen[fp] = true
|
||||
}
|
||||
}
|
||||
for i := range callers {
|
||||
fp := fmt.Sprintf("fp-race-%d", i)
|
||||
if !seen[fp] {
|
||||
t.Errorf("alert %q is missing: its delivery's whole payload was silently rolled back "+
|
||||
"when it lost the race for the incident", fp)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -15,6 +15,12 @@ const (
|
||||
// expiryGrace absorbs clock skew and notification latency before an alert
|
||||
// whose ends_at watermark has passed is treated as stale.
|
||||
expiryGrace = 5 * time.Minute
|
||||
|
||||
// archiverLockKey is the Postgres advisory lock the sweeper takes for the
|
||||
// duration of each pass, so that running more than one replica does not run
|
||||
// the sweep concurrently on all of them. Its value has no meaning beyond
|
||||
// being distinct from notifierLockKey.
|
||||
archiverLockKey int64 = 7265_0001
|
||||
)
|
||||
|
||||
// StartArchiver runs the alert sweeper until ctx is cancelled, starting with an
|
||||
@@ -23,15 +29,25 @@ const (
|
||||
// the fallback, not the setting: each pass reads the current value from the
|
||||
// settings table, so an administrator's change takes effect on the next tick
|
||||
// instead of at the next restart.
|
||||
//
|
||||
// Each pass runs under archiverLockKey (see withAdvisoryLock), so that on more
|
||||
// than one replica only whichever instance's tick takes the lock first actually
|
||||
// sweeps; the rest skip that tick rather than racing the same pass.
|
||||
func StartArchiver(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, notify NotifyConfig) {
|
||||
ticker := time.NewTicker(sweepInterval)
|
||||
defer ticker.Stop()
|
||||
|
||||
Sweep(ctx, db, archiveAfter, staleAfter, notify)
|
||||
sweep := func() {
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "sweeper", func() {
|
||||
Sweep(ctx, db, archiveAfter, staleAfter, notify)
|
||||
})
|
||||
}
|
||||
|
||||
sweep()
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
Sweep(ctx, db, archiveAfter, staleAfter, notify)
|
||||
sweep()
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
@@ -126,7 +142,7 @@ func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip
|
||||
continue
|
||||
}
|
||||
alertID := id
|
||||
if err := logEvent(ctx, db, incidentID, evAlertResolved, nil, &alertID, nil); err != nil {
|
||||
if err := logEvent(ctx, db, incidentID, evAlertResolved, nil, nil, &alertID, nil); err != nil {
|
||||
log.Printf("sweeper: log expiry event: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
+16
-5
@@ -99,13 +99,24 @@ func (c Caller) ServiceAccountID() (int64, bool) {
|
||||
return c.sa.id, true
|
||||
}
|
||||
|
||||
// ServiceAccountName reports this caller's own service-account name, for a
|
||||
// handler's synchronous response — the same credential it authenticated
|
||||
// with, already resolved onto the Caller by serveAsServiceAccount, so no
|
||||
// extra query is needed.
|
||||
func (c Caller) ServiceAccountName() (string, bool) {
|
||||
if c.sa == nil {
|
||||
return "", false
|
||||
}
|
||||
return c.sa.name, true
|
||||
}
|
||||
|
||||
// Identity is a stable, log/audit-facing string distinguishing a human
|
||||
// caller from a service account — "user:42" or "service-account:7". Not
|
||||
// wired into any database column today (incidents.go's acknowledged_by/
|
||||
// assigned_to/user_id are explicitly out of scope for this change — that
|
||||
// needs its own schema migration, tracked separately), but this is the one
|
||||
// place in the request path that already knows which kind of caller this
|
||||
// is, and that follow-up will want exactly this accessor.
|
||||
// wired into any database column — incidents.go's acknowledged_by/
|
||||
// incident_events.user_id use AsHuman()/ServiceAccountID() directly against
|
||||
// the parallel *_service_account_id columns (migration 015) instead, since a
|
||||
// column needs the id, not this rendered string. assigned_to stays
|
||||
// human-only and out of scope (terdut-server#25's follow-up).
|
||||
func (c Caller) Identity() string {
|
||||
switch {
|
||||
case c.user != nil:
|
||||
|
||||
@@ -373,7 +373,7 @@ func deadmanDied(ctx context.Context, db *sql.DB, notify NotifyConfig, hb deadma
|
||||
|
||||
alertID := hb.id
|
||||
detail := "last heartbeat " + humanDuration(now.Sub(time.Unix(hb.receivedAt, 0))) + " ago"
|
||||
if err := logEvent(ctx, tx, incidentID, evDeadmanSilent, nil, &alertID, &detail); err != nil {
|
||||
if err := logEvent(ctx, tx, incidentID, evDeadmanSilent, nil, nil, &alertID, &detail); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -415,7 +415,7 @@ func deadmanRecovered(ctx context.Context, db *sql.DB, hb deadmanAlert) error {
|
||||
time.Now().Unix(), incidentResolutionRecovered, incidentID); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := logEvent(ctx, tx, incidentID, evResolved, nil, nil, nil); err != nil {
|
||||
if err := logEvent(ctx, tx, incidentID, evResolved, nil, nil, nil, nil); err != nil {
|
||||
return err
|
||||
}
|
||||
// The all-clear goes to whoever was paged, which enqueueResolved works out
|
||||
|
||||
@@ -229,7 +229,7 @@ func advanceEscalation(ctx context.Context, db *sql.DB, cfg NotifyConfig, policy
|
||||
// nobody. That is a policy that looks configured and is not.
|
||||
detail += ": nobody reachable"
|
||||
}
|
||||
if err := logEvent(ctx, tx, incidentID, evEscalated, nil, nil, &detail); err != nil {
|
||||
if err := logEvent(ctx, tx, incidentID, evEscalated, nil, nil, nil, &detail); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
@@ -256,7 +256,7 @@ func escalationExhausted(ctx context.Context, tx *sql.Tx, policy *escalationPoli
|
||||
incidentID); err != nil {
|
||||
return err
|
||||
}
|
||||
return logEvent(ctx, tx, incidentID, evEscalated, nil, nil, &detail)
|
||||
return logEvent(ctx, tx, incidentID, evEscalated, nil, nil, nil, &detail)
|
||||
}
|
||||
|
||||
// pageLevel notifies every target of one level and reports who was woken.
|
||||
|
||||
@@ -1,8 +1,11 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -77,3 +80,39 @@ func decodeJSON(r *http.Request, v any) error {
|
||||
func errResp(msg string) map[string]string {
|
||||
return map[string]string{"error": msg}
|
||||
}
|
||||
|
||||
// withAdvisoryLock runs fn only if it can take the named Postgres advisory lock on a
|
||||
// dedicated connection, and skips fn otherwise. This is what keeps the archiver and
|
||||
// notifier safe to run on more than one replica: whichever instance's tick gets there
|
||||
// first does the work; the rest see the lock held and simply wait for their next tick
|
||||
// instead of running the same pass concurrently.
|
||||
//
|
||||
// pg_try_advisory_lock is session-scoped, so taking and releasing it must happen on the
|
||||
// same connection, reserved via db.Conn rather than borrowed from the pool's shared
|
||||
// connections fn itself may use — and released (unlocked, then closed) before returning,
|
||||
// since a session lock otherwise outlives this call and leaks onto whatever reuses the
|
||||
// pooled connection next.
|
||||
func withAdvisoryLock(ctx context.Context, db *sql.DB, key int64, name string, fn func()) {
|
||||
conn, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
log.Printf("%s: advisory lock: acquire connection: %v", name, err)
|
||||
return
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
var locked bool
|
||||
if err := conn.QueryRowContext(ctx, "SELECT pg_try_advisory_lock($1)", key).Scan(&locked); err != nil {
|
||||
log.Printf("%s: advisory lock: %v", name, err)
|
||||
return
|
||||
}
|
||||
if !locked {
|
||||
return // another replica is already running this pass
|
||||
}
|
||||
defer func() {
|
||||
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", key); err != nil {
|
||||
log.Printf("%s: advisory unlock: %v", name, err)
|
||||
}
|
||||
}()
|
||||
|
||||
fn()
|
||||
}
|
||||
|
||||
@@ -65,11 +65,13 @@ const incidentSelectFrom = `
|
||||
WHERE el.team_id = i.team_id AND el.position = i.escalation_level),
|
||||
i.triggered_at,
|
||||
i.acknowledged_by, i.acknowledged_at, ack.username,
|
||||
i.acknowledged_by_service_account_id, acksa.name,
|
||||
i.assigned_to, asg.username, i.snoozed_until,
|
||||
i.resolved_at, i.resolution_source, i.archived_at
|
||||
FROM incidents i
|
||||
JOIN teams t ON t.id = i.team_id
|
||||
LEFT JOIN users ack ON ack.id = i.acknowledged_by
|
||||
LEFT JOIN service_accounts acksa ON acksa.id = i.acknowledged_by_service_account_id
|
||||
LEFT JOIN users asg ON asg.id = i.assigned_to`
|
||||
|
||||
func scanIncident(s scanner) (models.Incident, error) {
|
||||
@@ -83,6 +85,7 @@ func scanIncident(s scanner) (models.Incident, error) {
|
||||
&i.EscalationLevel, &escalationDue,
|
||||
&triggeredAt,
|
||||
&i.AcknowledgedByID, &ackAt, &i.AcknowledgedByUser,
|
||||
&i.AcknowledgedByServiceAccountID, &i.AcknowledgedByServiceAccountName,
|
||||
&i.AssignedToID, &i.AssignedToUser, &snoozedUntil,
|
||||
&resolvedAt, &i.ResolutionSource, &archivedAt,
|
||||
); err != nil {
|
||||
@@ -112,13 +115,31 @@ func fetchIncident(ctx context.Context, q querier, id int64) (models.Incident, e
|
||||
return scanIncident(q.QueryRowContext(ctx, incidentSelectFrom+" WHERE i.id = $1", id))
|
||||
}
|
||||
|
||||
// logEvent appends one entry to an incident's timeline. A nil userID means the
|
||||
// server acted rather than a person.
|
||||
func logEvent(ctx context.Context, q querier, incidentID int64, evType string, userID, alertID *int64, detail *string) error {
|
||||
// callerActorIDs resolves the current request's caller into the pair of
|
||||
// nilable ids logEvent/acknowledgeIncidentAs expect: exactly one of userID/
|
||||
// serviceAccountID is set (never both), replacing the unchecked
|
||||
// userFromContext(ctx) zero-value reads that used to write a human-only id
|
||||
// of 0 for a service-account caller (terdut-server#25).
|
||||
func callerActorIDs(ctx context.Context) (userID, serviceAccountID *int64) {
|
||||
caller, _ := callerFromContext(ctx)
|
||||
if u, ok := caller.AsHuman(); ok {
|
||||
return &u.ID, nil
|
||||
}
|
||||
if id, ok := caller.ServiceAccountID(); ok {
|
||||
return nil, &id
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
// logEvent appends one entry to an incident's timeline. userID and
|
||||
// serviceAccountID are mutually exclusive and both nilable; both nil means
|
||||
// the server acted rather than any caller (see incident_events_actor_xor_chk,
|
||||
// migration 015).
|
||||
func logEvent(ctx context.Context, q querier, incidentID int64, evType string, userID, serviceAccountID, alertID *int64, detail *string) error {
|
||||
_, err := q.ExecContext(ctx, `
|
||||
INSERT INTO incident_events (incident_id, type, user_id, alert_id, detail, created_at)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
incidentID, evType, userID, alertID, detail, time.Now().Unix())
|
||||
INSERT INTO incident_events (incident_id, type, user_id, service_account_id, alert_id, detail, created_at)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7)`,
|
||||
incidentID, evType, userID, serviceAccountID, alertID, detail, time.Now().Unix())
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -251,7 +272,7 @@ func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, e
|
||||
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
||||
return false, err
|
||||
}
|
||||
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil); err != nil {
|
||||
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil, nil); err != nil {
|
||||
return false, err
|
||||
}
|
||||
// The all-clear goes only to whoever was paged in the first place, which
|
||||
@@ -260,19 +281,30 @@ func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, e
|
||||
return true, enqueueResolved(ctx, q, incidentID)
|
||||
}
|
||||
|
||||
// acknowledgeIncident records that userID has picked an incident up, and reports
|
||||
// whether it changed anything — an already-resolved or already-acknowledged
|
||||
// incident is left alone, so a second acknowledge (a retried request, or a
|
||||
// stale push notification tapped after the web UI already acked it) is a
|
||||
// no-op rather than a second "acknowledged" timeline entry. Shared by the
|
||||
// authenticated handler and the Acknowledge button in a push notification,
|
||||
// so both write the same state and the same timeline entry.
|
||||
// acknowledgeIncident records that userID — a human — has picked an incident
|
||||
// up, and reports whether it changed anything — an already-resolved or
|
||||
// already-acknowledged incident is left alone, so a second acknowledge (a
|
||||
// retried request, or a stale push notification tapped after the web UI
|
||||
// already acked it) is a no-op rather than a second "acknowledged" timeline
|
||||
// entry. Used only by the Acknowledge button in a push notification
|
||||
// (notify_ack.go), which always resolves a human from
|
||||
// incident_ack_tokens.user_id — there is no service-account equivalent of
|
||||
// that flow, so this keeps its human-only signature; the authenticated
|
||||
// handler goes through acknowledgeIncidentAs below instead.
|
||||
func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int64) (bool, error) {
|
||||
return acknowledgeIncidentAs(ctx, q, incidentID, &userID, nil)
|
||||
}
|
||||
|
||||
// acknowledgeIncidentAs is acknowledgeIncident generalized to either actor
|
||||
// kind. userID and serviceAccountID are mutually exclusive and nilable the
|
||||
// same way logEvent's are (see incidents_ack_actor_xor_chk, migration 015).
|
||||
func acknowledgeIncidentAs(ctx context.Context, q querier, incidentID int64, userID, serviceAccountID *int64) (bool, error) {
|
||||
res, err := q.ExecContext(ctx, `
|
||||
UPDATE incidents
|
||||
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_at = $2
|
||||
WHERE id = $3 AND status = 'triggered'`,
|
||||
userID, time.Now().Unix(), incidentID)
|
||||
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_by_service_account_id = $2,
|
||||
acknowledged_at = $3
|
||||
WHERE id = $4 AND status = 'triggered'`,
|
||||
userID, serviceAccountID, time.Now().Unix(), incidentID)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
@@ -283,7 +315,7 @@ func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int6
|
||||
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return true, logEvent(ctx, q, incidentID, evAcknowledged, &userID, nil, nil)
|
||||
return true, logEvent(ctx, q, incidentID, evAcknowledged, userID, serviceAccountID, nil, nil)
|
||||
}
|
||||
|
||||
// openIncidentForAlert returns the open incident an alert currently belongs to,
|
||||
|
||||
+34
-24
@@ -157,9 +157,11 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
rows, err := db.QueryContext(r.Context(), `
|
||||
SELECT e.id, e.incident_id, e.type, e.user_id, u.username,
|
||||
e.service_account_id, sa.name,
|
||||
e.alert_id, e.detail, e.created_at
|
||||
FROM incident_events e
|
||||
LEFT JOIN users u ON u.id = e.user_id
|
||||
LEFT JOIN service_accounts sa ON sa.id = e.service_account_id
|
||||
WHERE e.incident_id = $1
|
||||
ORDER BY e.created_at ASC, e.id ASC`, id)
|
||||
if err != nil {
|
||||
@@ -173,6 +175,7 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
||||
var e models.IncidentEvent
|
||||
var ts int64
|
||||
if err := rows.Scan(&e.ID, &e.IncidentID, &e.Type, &e.UserID, &e.Username,
|
||||
&e.ServiceAccountID, &e.ServiceAccountName,
|
||||
&e.AlertID, &e.Detail, &ts); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -190,8 +193,8 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
user, _ := userFromContext(r.Context())
|
||||
acked, err := acknowledgeIncident(r.Context(), db, id, user.ID)
|
||||
userID, saID := callerActorIDs(r.Context())
|
||||
acked, err := acknowledgeIncidentAs(r.Context(), db, id, userID, saID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -223,13 +226,14 @@ func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc {
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
user, _ := userFromContext(r.Context())
|
||||
userID, saID := callerActorIDs(r.Context())
|
||||
if !updateOpenIncident(w, r, db, id,
|
||||
`UPDATE incidents SET status = 'triggered', acknowledged_by = NULL, acknowledged_at = NULL
|
||||
`UPDATE incidents SET status = 'triggered', acknowledged_by = NULL,
|
||||
acknowledged_by_service_account_id = NULL, acknowledged_at = NULL
|
||||
WHERE id = $1 AND resolved_at IS NULL`, id) {
|
||||
return
|
||||
}
|
||||
if err := logEvent(r.Context(), db, id, evUnacknowledged, &user.ID, nil, nil); err != nil {
|
||||
if err := logEvent(r.Context(), db, id, evUnacknowledged, userID, saID, nil, nil); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
@@ -247,7 +251,7 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc {
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
user, _ := userFromContext(r.Context())
|
||||
userID, saID := callerActorIDs(r.Context())
|
||||
// The body is optional: clients that predate resolution notes send none.
|
||||
var req struct {
|
||||
Resolution string `json:"resolution"`
|
||||
@@ -268,12 +272,12 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if err := logEvent(r.Context(), db, id, evResolved, &user.ID, nil, nil); err != nil {
|
||||
if err := logEvent(r.Context(), db, id, evResolved, userID, saID, nil, nil); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if req.Resolution != "" {
|
||||
if err := logEvent(r.Context(), db, id, evResolutionNote, &user.ID, nil, &req.Resolution); err != nil {
|
||||
if err := logEvent(r.Context(), db, id, evResolutionNote, userID, saID, nil, &req.Resolution); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
@@ -312,7 +316,7 @@ func handleIncidentAssign(db *sql.DB) http.HandlerFunc {
|
||||
return
|
||||
}
|
||||
// On an "assigned" event user_id is the assignee, not the actor.
|
||||
if err := logEvent(r.Context(), db, id, evAssigned, &req.UserID, nil, nil); err != nil {
|
||||
if err := logEvent(r.Context(), db, id, evAssigned, &req.UserID, nil, nil, nil); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
@@ -363,14 +367,14 @@ func handleIncidentSnooze(db *sql.DB) http.HandlerFunc {
|
||||
return
|
||||
}
|
||||
|
||||
user, _ := userFromContext(r.Context())
|
||||
userID, saID := callerActorIDs(r.Context())
|
||||
if !updateOpenIncident(w, r, db, id,
|
||||
"UPDATE incidents SET snoozed_until = $1 WHERE id = $2 AND resolved_at IS NULL",
|
||||
until.Unix(), id) {
|
||||
return
|
||||
}
|
||||
detail := until.UTC().Format(time.RFC3339)
|
||||
if err := logEvent(r.Context(), db, id, evSnoozed, &user.ID, nil, &detail); err != nil {
|
||||
if err := logEvent(r.Context(), db, id, evSnoozed, userID, saID, nil, &detail); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
@@ -384,12 +388,12 @@ func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc {
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
user, _ := userFromContext(r.Context())
|
||||
userID, saID := callerActorIDs(r.Context())
|
||||
if !updateOpenIncident(w, r, db, id,
|
||||
"UPDATE incidents SET snoozed_until = NULL WHERE id = $1 AND resolved_at IS NULL", id) {
|
||||
return
|
||||
}
|
||||
if err := logEvent(r.Context(), db, id, evUnsnoozed, &user.ID, nil, nil); err != nil {
|
||||
if err := logEvent(r.Context(), db, id, evUnsnoozed, userID, saID, nil, nil); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
@@ -466,27 +470,32 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
||||
return
|
||||
}
|
||||
|
||||
user, _ := userFromContext(r.Context())
|
||||
caller, _ := callerFromContext(r.Context())
|
||||
userID, saID := callerActorIDs(r.Context())
|
||||
now := time.Now()
|
||||
var eventID int64
|
||||
err := db.QueryRowContext(r.Context(), `
|
||||
INSERT INTO incident_events (incident_id, type, user_id, detail, created_at)
|
||||
VALUES ($1, $2, $3, $4, $5)
|
||||
RETURNING id`, id, noteType, user.ID, req.Content, now.Unix()).Scan(&eventID)
|
||||
INSERT INTO incident_events (incident_id, type, user_id, service_account_id, detail, created_at)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)
|
||||
RETURNING id`, id, noteType, userID, saID, req.Content, now.Unix()).Scan(&eventID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
respond(w, http.StatusCreated, models.IncidentEvent{
|
||||
resp := models.IncidentEvent{
|
||||
ID: eventID,
|
||||
IncidentID: id,
|
||||
Type: noteType,
|
||||
UserID: &user.ID,
|
||||
Username: &user.Username,
|
||||
Detail: &req.Content,
|
||||
CreatedAt: now.UTC().Truncate(time.Second),
|
||||
})
|
||||
}
|
||||
if u, ok := caller.AsHuman(); ok {
|
||||
resp.UserID, resp.Username = &u.ID, &u.Username
|
||||
} else if saName, ok := caller.ServiceAccountName(); ok {
|
||||
resp.ServiceAccountID, resp.ServiceAccountName = saID, &saName
|
||||
}
|
||||
respond(w, http.StatusCreated, resp)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -504,11 +513,12 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc {
|
||||
return
|
||||
}
|
||||
|
||||
user, _ := userFromContext(r.Context())
|
||||
userID, saID := callerActorIDs(r.Context())
|
||||
res, err := db.ExecContext(r.Context(), `
|
||||
DELETE FROM incident_events
|
||||
WHERE id = $1 AND incident_id = $2 AND type IN ($3, $4) AND user_id = $5`,
|
||||
eventID, id, evNote, evResolutionNote, user.ID)
|
||||
WHERE id = $1 AND incident_id = $2 AND type IN ($3, $4)
|
||||
AND (user_id = $5 OR service_account_id = $6)`,
|
||||
eventID, id, evNote, evResolutionNote, userID, saID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
)
|
||||
|
||||
// amAlert builds one alert of a webhook payload.
|
||||
@@ -642,6 +643,106 @@ func TestIncident_ArchiveRoundTrip(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Service accounts (terdut-server#25)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestServiceAccount_CanActOnItsTeamsIncidents is #25's regression test.
|
||||
// Before the fix: acknowledge/resolve/snooze/create-note each 500'd (writing
|
||||
// acknowledged_by/user_id = 0, violating the users(id) FK for a service
|
||||
// account), and delete-note silently matched zero rows (WHERE user_id = 0)
|
||||
// instead of deleting.
|
||||
func TestServiceAccount_CanActOnItsTeamsIncidents(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
var integration struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/integrations",
|
||||
map[string]string{"name": "test"}), &integration)
|
||||
postToIntegration(t, s, integration.Key, "fp-sa", "SAIncident") // incident 1
|
||||
|
||||
// Acknowledge.
|
||||
resp := s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/acknowledge", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("service account acknowledge: %d", resp.StatusCode)
|
||||
}
|
||||
var inc map[string]any
|
||||
decode(t, resp, &inc)
|
||||
if inc["acknowledged_by_service_account_id"] == nil {
|
||||
t.Error("expected acknowledged_by_service_account_id to be set")
|
||||
}
|
||||
if inc["acknowledged_by_id"] != nil {
|
||||
t.Errorf("expected acknowledged_by_id to stay nil for a service-account actor, got %v", inc["acknowledged_by_id"])
|
||||
}
|
||||
|
||||
// Unacknowledge.
|
||||
resp = s.reqAs(t, keyA, http.MethodDelete, "/api/incidents/1/acknowledge", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("service account unacknowledge: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// Snooze, then unsnooze.
|
||||
resp = s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/snooze",
|
||||
map[string]string{"duration": "1h"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("service account snooze: %d", resp.StatusCode)
|
||||
}
|
||||
resp = s.reqAs(t, keyA, http.MethodDelete, "/api/incidents/1/snooze", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("service account unsnooze: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// Create, then delete, a note.
|
||||
var note map[string]any
|
||||
decode(t, s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/notes",
|
||||
map[string]string{"content": "looking into it"}), ¬e)
|
||||
if note["service_account_id"] == nil {
|
||||
t.Error("expected service_account_id on the note event")
|
||||
}
|
||||
if note["user_id"] != nil {
|
||||
t.Errorf("expected no user_id on a service-account note, got %v", note["user_id"])
|
||||
}
|
||||
noteID := int(note["id"].(float64))
|
||||
delResp := s.reqAs(t, keyA, http.MethodDelete, fmt.Sprintf("/api/incidents/1/notes/%d", noteID), nil)
|
||||
delResp.Body.Close()
|
||||
if delResp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("service account deleting its own note: %d", delResp.StatusCode)
|
||||
}
|
||||
|
||||
// Resolve.
|
||||
resp = s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/resolve", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("service account resolve: %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// Regression guard: a human actor must still write only the human columns,
|
||||
// unaffected by the service-account branch added above.
|
||||
func TestIncident_AcknowledgeStillWritesOnlyHumanColumn(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-human-ack", "Z", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
|
||||
var inc map[string]any
|
||||
decode(t, s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil), &inc)
|
||||
if inc["acknowledged_by_id"] == nil {
|
||||
t.Error("expected acknowledged_by_id to be set for a human actor")
|
||||
}
|
||||
if inc["acknowledged_by_service_account_id"] != nil {
|
||||
t.Errorf("expected acknowledged_by_service_account_id to stay nil for a human actor, got %v",
|
||||
inc["acknowledged_by_service_account_id"])
|
||||
}
|
||||
}
|
||||
|
||||
func TestSweeper_ArchivesResolvedIncidents(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postWebhook(t, s, []map[string]any{
|
||||
|
||||
@@ -38,6 +38,13 @@ const (
|
||||
ackTokenTTL = 24 * time.Hour
|
||||
)
|
||||
|
||||
// notifierLockKey is the Postgres advisory lock the notifier takes for the
|
||||
// duration of each pass, so that running more than one replica does not
|
||||
// deliver (or double-deliver) the same notification from more than one of
|
||||
// them at once. Its value has no meaning beyond being distinct from
|
||||
// archiverLockKey.
|
||||
const notifierLockKey int64 = 7265_0002
|
||||
|
||||
// Notification kinds, recording why a push was sent.
|
||||
const (
|
||||
notifyTriggered = "triggered"
|
||||
@@ -97,6 +104,10 @@ var notifyClient = &http.Client{Timeout: 10 * time.Second}
|
||||
|
||||
// StartNotifier delivers queued notifications until ctx is cancelled, starting
|
||||
// with an immediate pass so a restart flushes whatever the last one left behind.
|
||||
//
|
||||
// Each pass runs under notifierLockKey (see withAdvisoryLock), so that on more
|
||||
// than one replica only whichever instance's tick takes the lock first actually
|
||||
// delivers; the rest skip that tick rather than racing the same pass.
|
||||
func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
if !cfg.enabled() {
|
||||
log.Print("notifier: disabled (no ntfy URL configured)")
|
||||
@@ -107,11 +118,17 @@ func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
ticker := time.NewTicker(notifyInterval)
|
||||
defer ticker.Stop()
|
||||
|
||||
NotifySweep(ctx, db, cfg)
|
||||
sweep := func() {
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "notifier", func() {
|
||||
NotifySweep(ctx, db, cfg)
|
||||
})
|
||||
}
|
||||
|
||||
sweep()
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
NotifySweep(ctx, db, cfg)
|
||||
sweep()
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
@@ -240,7 +257,7 @@ func deliverPending(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
}
|
||||
// Logged, not returned: the page has already gone out, and treating a
|
||||
// failed timeline write as a failed delivery would send it again.
|
||||
if err := logEvent(ctx, db, n.incidentID, eventNotified, n.userID, nil, &n.kind); err != nil {
|
||||
if err := logEvent(ctx, db, n.incidentID, eventNotified, n.userID, nil, nil, &n.kind); err != nil {
|
||||
log.Printf("notifier: log delivery of %d: %v", n.id, err)
|
||||
}
|
||||
sent++
|
||||
@@ -295,7 +312,7 @@ func markFailed(ctx context.Context, db *sql.DB, n outboxRow, cause error) {
|
||||
return
|
||||
}
|
||||
detail := fmt.Sprintf("%s: %s", n.kind, cause)
|
||||
if err := logEvent(ctx, db, n.incidentID, eventNotifyFailed, n.userID, nil, &detail); err != nil {
|
||||
if err := logEvent(ctx, db, n.incidentID, eventNotifyFailed, n.userID, nil, nil, &detail); err != nil {
|
||||
log.Printf("notifier: log failure of %d: %v", n.id, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"embed"
|
||||
"fmt"
|
||||
@@ -62,6 +63,16 @@ func Open(dsn string) (*sql.DB, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// migrationLockKey is the Postgres advisory lock Migrate holds for its whole
|
||||
// run. Two replicas starting at once would otherwise race the check-then-apply
|
||||
// loop below against schema_migrations: the loser could crash on a
|
||||
// duplicate-key insert, or contend with the winner's uncommitted DDL. Blocking
|
||||
// (pg_advisory_lock, not pg_try_advisory_lock as the archiver and notifier
|
||||
// use): on boot there is no later tick to defer to, so the right behaviour is
|
||||
// to wait for the other replica to finish migrating, not to skip ahead and
|
||||
// start serving against an unmigrated schema.
|
||||
const migrationLockKey int64 = 7265_0003
|
||||
|
||||
// Migrate applies every embedded migration that has not been applied yet, in
|
||||
// filename order, recording each in schema_migrations.
|
||||
//
|
||||
@@ -69,6 +80,22 @@ func Open(dsn string) (*sql.DB, error) {
|
||||
// migration that failed half way used to leave the schema in whatever state it
|
||||
// had reached. Postgres has transactional DDL, so the rollback is real.
|
||||
func Migrate(db *sql.DB) error {
|
||||
ctx := context.Background()
|
||||
conn, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("migrate: acquire connection: %w", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_lock($1)", migrationLockKey); err != nil {
|
||||
return fmt.Errorf("migrate: acquire advisory lock: %w", err)
|
||||
}
|
||||
defer func() {
|
||||
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", migrationLockKey); err != nil {
|
||||
log.Printf("migrate: release advisory lock: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := db.Exec(`CREATE TABLE IF NOT EXISTS schema_migrations (
|
||||
version TEXT PRIMARY KEY,
|
||||
applied_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
package db_test
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
||||
|
||||
_ "github.com/jackc/pgx/v5/stdlib"
|
||||
)
|
||||
|
||||
// TERDUT_TEST_DSN must point at a database the test role may create schemas
|
||||
// in; see internal/api/testdb_test.go for the fuller rationale this mirrors.
|
||||
// An unset DSN fails rather than skips, deliberately.
|
||||
const testDSNEnv = "TERDUT_TEST_DSN"
|
||||
|
||||
// TestMigrate_ConcurrentCallersDoNotRace reproduces two replicas starting at
|
||||
// once against a brand-new, unmigrated schema: both call db.Migrate at the
|
||||
// same time. Before migrationLockKey, the loser could crash on a
|
||||
// duplicate-key insert into schema_migrations, or contend with the winner's
|
||||
// uncommitted DDL; with the advisory lock, one blocks until the other
|
||||
// finishes and both return cleanly.
|
||||
func TestMigrate_ConcurrentCallersDoNotRace(t *testing.T) {
|
||||
dsn := os.Getenv(testDSNEnv)
|
||||
if dsn == "" {
|
||||
t.Fatalf("%s is not set: these tests need Postgres.\n"+
|
||||
"Run `make test-db` for a local one, then\n"+
|
||||
" export %s=postgres://terdut:terdut@localhost:5432/terdut_test?sslmode=disable",
|
||||
testDSNEnv, testDSNEnv)
|
||||
}
|
||||
|
||||
schema := fmt.Sprintf("migrate_race_%d", os.Getpid())
|
||||
admin, err := sql.Open("pgx", dsn)
|
||||
if err != nil {
|
||||
t.Fatalf("connect to %s: %v", testDSNEnv, err)
|
||||
}
|
||||
defer admin.Close()
|
||||
if _, err := admin.Exec("CREATE SCHEMA " + schema); err != nil {
|
||||
t.Fatalf("create schema %s: %v", schema, err)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
cleanup, err := sql.Open("pgx", dsn)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
defer cleanup.Close()
|
||||
if _, err := cleanup.Exec("DROP SCHEMA " + schema + " CASCADE"); err != nil {
|
||||
t.Logf("drop schema %s: %v", schema, err)
|
||||
}
|
||||
})
|
||||
|
||||
scoped := withSearchPath(dsn, schema)
|
||||
|
||||
const callers = 2
|
||||
errs := make([]error, callers)
|
||||
var wg sync.WaitGroup
|
||||
for i := range callers {
|
||||
wg.Add(1)
|
||||
go func(i int) {
|
||||
defer wg.Done()
|
||||
database, err := db.Open(scoped)
|
||||
if err != nil {
|
||||
errs[i] = fmt.Errorf("open: %w", err)
|
||||
return
|
||||
}
|
||||
defer database.Close()
|
||||
errs[i] = db.Migrate(database)
|
||||
}(i)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
for i, err := range errs {
|
||||
if err != nil {
|
||||
t.Fatalf("Migrate #%d: %v", i, err)
|
||||
}
|
||||
}
|
||||
|
||||
entries, err := os.ReadDir("migrations")
|
||||
if err != nil {
|
||||
t.Fatalf("read migrations dir: %v", err)
|
||||
}
|
||||
var want int
|
||||
for _, e := range entries {
|
||||
if !e.IsDir() && strings.HasSuffix(e.Name(), ".sql") {
|
||||
want++
|
||||
}
|
||||
}
|
||||
|
||||
check, err := sql.Open("pgx", scoped)
|
||||
if err != nil {
|
||||
t.Fatalf("connect for verification: %v", err)
|
||||
}
|
||||
defer check.Close()
|
||||
|
||||
var got int
|
||||
if err := check.QueryRow("SELECT COUNT(*) FROM schema_migrations").Scan(&got); err != nil {
|
||||
t.Fatalf("count schema_migrations: %v", err)
|
||||
}
|
||||
if got != want {
|
||||
t.Fatalf("schema_migrations has %d row(s) after two concurrent Migrate calls, want %d (one per migration file, no duplicates)", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// withSearchPath pins a DSN to one schema. Copied from
|
||||
// internal/api/testdb_test.go rather than shared: that helper lives in the
|
||||
// api_test package, a separate compiled package this one cannot import.
|
||||
func withSearchPath(dsn, schema string) string {
|
||||
opt := "-csearch_path=" + schema
|
||||
|
||||
if strings.HasPrefix(dsn, "postgres://") || strings.HasPrefix(dsn, "postgresql://") {
|
||||
u, err := url.Parse(dsn)
|
||||
if err == nil {
|
||||
q := u.Query()
|
||||
q.Set("options", opt)
|
||||
u.RawQuery = q.Encode()
|
||||
return u.String()
|
||||
}
|
||||
}
|
||||
return dsn + " options='" + opt + "'"
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
-- Service-account actors on incident mutations (terdut-server#25). A
|
||||
-- team-scoped service account acknowledging/resolving/snoozing/noting an
|
||||
-- incident is not a users row, so it cannot be written into
|
||||
-- acknowledged_by/incident_events.user_id — doing so either violates the
|
||||
-- users(id) FK (new rows) or, for incident_events.user_id, silently matches
|
||||
-- zero rows on delete. These columns are the service-account-shaped parallel
|
||||
-- to the existing human ones: nullable, mutually exclusive with their human
|
||||
-- counterpart, ON DELETE SET NULL so a deleted service account doesn't take
|
||||
-- the incident history with it.
|
||||
ALTER TABLE incidents
|
||||
ADD COLUMN acknowledged_by_service_account_id BIGINT
|
||||
REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||
|
||||
ALTER TABLE incident_events
|
||||
ADD COLUMN service_account_id BIGINT
|
||||
REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||
|
||||
-- At most one actor kind per row: both NULL ("the server acted") is valid,
|
||||
-- exactly one set is valid, both set is a bug this constraint refuses to
|
||||
-- store rather than silently accepting.
|
||||
ALTER TABLE incidents
|
||||
ADD CONSTRAINT incidents_ack_actor_xor_chk CHECK (
|
||||
acknowledged_by IS NULL OR acknowledged_by_service_account_id IS NULL
|
||||
);
|
||||
|
||||
ALTER TABLE incident_events
|
||||
ADD CONSTRAINT incident_events_actor_xor_chk CHECK (
|
||||
user_id IS NULL OR service_account_id IS NULL
|
||||
);
|
||||
|
||||
CREATE INDEX incidents_acknowledged_by_service_account_id_idx
|
||||
ON incidents(acknowledged_by_service_account_id);
|
||||
CREATE INDEX incident_events_service_account_id_idx
|
||||
ON incident_events(service_account_id);
|
||||
|
||||
-- assigned_to_service_account_id is deliberately not added here: it would sit
|
||||
-- unpopulated until handleIncidentAssign itself tracks an actor, which is a
|
||||
-- separate, pre-existing gap (it records the assignee today, never the
|
||||
-- actor, for humans either) tracked in its own follow-up issue.
|
||||
+27
-10
@@ -43,6 +43,13 @@ type Incident struct {
|
||||
AcknowledgedByUser *string `json:"acknowledged_by,omitempty"`
|
||||
AcknowledgedAt *time.Time `json:"acknowledged_at,omitempty"`
|
||||
|
||||
// AcknowledgedByServiceAccountID/Name are the service-account-shaped
|
||||
// parallel to AcknowledgedByID/User above: mutually exclusive with it,
|
||||
// populated when a service account (not a human) acknowledged this
|
||||
// incident. See migration 015 and terdut-server#25.
|
||||
AcknowledgedByServiceAccountID *int64 `json:"acknowledged_by_service_account_id,omitempty"`
|
||||
AcknowledgedByServiceAccountName *string `json:"acknowledged_by_service_account,omitempty"`
|
||||
|
||||
AssignedToID *int64 `json:"assigned_to_id,omitempty"`
|
||||
AssignedToUser *string `json:"assigned_to,omitempty"`
|
||||
|
||||
@@ -67,17 +74,27 @@ type Incident struct {
|
||||
// and is the only history this server keeps — alert rows are mutated in place.
|
||||
//
|
||||
// Type is one of: triggered, alert_added, alert_resolved, acknowledged,
|
||||
// unacknowledged, assigned, snoozed, unsnoozed, resolved, note. A nil UserID
|
||||
// means the server acted rather than a person.
|
||||
// unacknowledged, assigned, snoozed, unsnoozed, resolved, note. UserID and
|
||||
// ServiceAccountID are mutually exclusive; both nil means the server acted
|
||||
// rather than any caller.
|
||||
type IncidentEvent struct {
|
||||
ID int64 `json:"id"`
|
||||
IncidentID int64 `json:"incident_id"`
|
||||
Type string `json:"type"`
|
||||
UserID *int64 `json:"user_id,omitempty"`
|
||||
Username *string `json:"username,omitempty"`
|
||||
AlertID *int64 `json:"alert_id,omitempty"`
|
||||
Detail *string `json:"detail,omitempty"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
ID int64 `json:"id"`
|
||||
IncidentID int64 `json:"incident_id"`
|
||||
Type string `json:"type"`
|
||||
UserID *int64 `json:"user_id,omitempty"`
|
||||
Username *string `json:"username,omitempty"`
|
||||
|
||||
// ServiceAccountID/Name are the service-account-shaped parallel to
|
||||
// UserID/Username above: mutually exclusive with it, populated when a
|
||||
// service account (not a human, and not nil-meaning-the-server-acted)
|
||||
// performed this event. Named Name, not Username — a ServiceAccount has
|
||||
// a Name field, not a Username. See migration 015 and terdut-server#25.
|
||||
ServiceAccountID *int64 `json:"service_account_id,omitempty"`
|
||||
ServiceAccountName *string `json:"service_account_name,omitempty"`
|
||||
|
||||
AlertID *int64 `json:"alert_id,omitempty"`
|
||||
Detail *string `json:"detail,omitempty"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
}
|
||||
|
||||
// SimilarIncident is an earlier, resolved incident with the same signature as
|
||||
|
||||
@@ -326,6 +326,7 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
/* ---------- chips ---------- */
|
||||
|
||||
.chips {
|
||||
position: relative;
|
||||
display: flex; gap: 6px;
|
||||
padding: 12px 16px 8px;
|
||||
overflow-x: auto; scrollbar-width: none;
|
||||
@@ -341,11 +342,14 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
}
|
||||
.chip[aria-selected="true"] { background: var(--text); border-color: var(--text); color: var(--bg); }
|
||||
.chip .count { margin-left: 4px; opacity: 0.7; }
|
||||
/* Pinned to the visible right edge of the scrolling row (sticky, not
|
||||
absolute, so it tracks the scroll position rather than the content). */
|
||||
/* An overlay, not a flex item: absolute against .chips' own (non-scrolling)
|
||||
box stays flush with its real right edge regardless of scroll position,
|
||||
which turned out not to be true of position:sticky here — as a flex
|
||||
item, its sticky offset interacted with the row's gap and its own
|
||||
negative margin, landing short of the edge by about one gap's width. */
|
||||
.chips-fade {
|
||||
position: sticky; right: -1px; flex: none;
|
||||
width: 24px; margin-left: -24px;
|
||||
position: absolute; top: 0; right: 0; bottom: 0;
|
||||
width: 24px;
|
||||
background: linear-gradient(to right, transparent, var(--bg));
|
||||
pointer-events: none;
|
||||
}
|
||||
@@ -579,6 +583,10 @@ details[open] > summary { margin-bottom: 8px; }
|
||||
}
|
||||
.actionbar .btn { min-height: 48px; }
|
||||
.actionbar .btn-primary { flex: 1; font-size: 16px; }
|
||||
/* The extra actions that only phones hide behind "More" — see actionBar() and
|
||||
extraActions() in incident.js. Hidden by default; the desktop block below
|
||||
shows them and hides the now-redundant More button instead. */
|
||||
.action-extra { display: none; }
|
||||
|
||||
.detail-placeholder {
|
||||
display: grid; place-items: center; height: 100%;
|
||||
@@ -665,7 +673,7 @@ details[open] > summary { margin-bottom: 8px; }
|
||||
caught by the unrelated .label > span styling meant for label chips. */
|
||||
.week-nav .week-label { font-size: 14px; font-weight: 650; min-width: 11.5em; text-align: center; white-space: nowrap; }
|
||||
.week-label small { color: var(--faint); font-weight: 600; font-size: 11px; margin-left: 2px; }
|
||||
.days { list-style: none; margin: 0; padding: 0; }
|
||||
.days { list-style: none; margin: 0 auto; padding: 0; }
|
||||
.day { display: grid; grid-template-columns: 3.2em 4.2em 1fr; align-items: center; gap: 8px; min-height: 50px; padding: 0 14px; }
|
||||
.day + .day { border-top: 1px solid var(--border); }
|
||||
.day-name { font-weight: 650; }
|
||||
@@ -769,7 +777,15 @@ kbd {
|
||||
.pane-list .chips { flex-wrap: wrap; overflow-x: visible; }
|
||||
.pane-list .chip-sep { display: none; }
|
||||
.chips-fade { display: none; }
|
||||
.view-queue:not(.has-detail) .pane-detail { display: block; }
|
||||
/* With nothing selected there is no detail to show next to, so the list
|
||||
takes the whole row instead of leaving the second column as dead space
|
||||
around the placeholder text. Selecting an incident (.has-detail) drops
|
||||
back to the base minmax(340,420) 1fr rule above. */
|
||||
.view-queue:not(.has-detail) { grid-template-columns: 1fr; }
|
||||
.view-queue:not(.has-detail) .pane-list { border-right: 0; }
|
||||
/* Full width reads better capped than edge-to-edge on a very wide monitor,
|
||||
matching .detail's own cap below. */
|
||||
.view-queue:not(.has-detail) .list { max-width: 900px; margin: 0 auto; }
|
||||
|
||||
/* On desktop the list stays visible next to the detail. */
|
||||
.app.detail-open .nav { display: flex; }
|
||||
@@ -783,8 +799,12 @@ kbd {
|
||||
.actionbar {
|
||||
position: sticky; bottom: 0;
|
||||
padding: 12px 32px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.actionbar .btn-primary { flex: 0 1 240px; }
|
||||
/* Room enough to show every action, so More is fully redundant here. */
|
||||
.action-extra { display: inline-flex; }
|
||||
.more-btn { display: none; }
|
||||
|
||||
.sheet {
|
||||
width: min(440px, calc(100% - 32px));
|
||||
|
||||
@@ -33,7 +33,9 @@ function render() {
|
||||
h('div', { class: 'card' }, shortcuts())),
|
||||
|
||||
h('div', { class: 'page-head' }),
|
||||
h('button', { class: 'btn btn-block', type: 'button', onclick: signOut }, icon('logout'), 'Sign out'),
|
||||
// Wrapped in a div: .btn is inline-flex, and only a block-level element
|
||||
// picks up .view-page > *'s margin:auto centering (see app.css:316).
|
||||
h('div', {}, h('button', { class: 'btn btn-block', type: 'button', onclick: signOut }, icon('logout'), 'Sign out')),
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -96,7 +96,9 @@ function render() {
|
||||
}
|
||||
|
||||
function backLink() {
|
||||
return h('a', { class: 'back-link', href: '/admin/teams' }, icon('chevronLeft'), h('span', { text: 'Teams' }));
|
||||
// Wrapped in a div: .back-link is inline-flex, and only a block-level
|
||||
// element picks up .view-page > *'s margin:auto centering (app.css:316).
|
||||
return h('div', {}, h('a', { class: 'back-link', href: '/admin/teams' }, icon('chevronLeft'), h('span', { text: 'Teams' })));
|
||||
}
|
||||
|
||||
// --- identity --------------------------------------------------------------
|
||||
|
||||
@@ -83,7 +83,9 @@ function render() {
|
||||
}
|
||||
|
||||
function backLink() {
|
||||
return h('a', { class: 'back-link', href: '/admin/users' }, icon('chevronLeft'), h('span', { text: 'Users' }));
|
||||
// Wrapped in a div: .back-link is inline-flex, and only a block-level
|
||||
// element picks up .view-page > *'s margin:auto centering (app.css:316).
|
||||
return h('div', {}, h('a', { class: 'back-link', href: '/admin/users' }, icon('chevronLeft'), h('span', { text: 'Users' })));
|
||||
}
|
||||
|
||||
// --- identity --------------------------------------------------------------
|
||||
|
||||
@@ -123,6 +123,16 @@ function who(id, name) {
|
||||
return name || 'someone';
|
||||
}
|
||||
|
||||
// ackActorLabel renders whoever acknowledged inc, human or service account —
|
||||
// the two are mutually exclusive (migration 015), and a service account is a
|
||||
// credential, not "you" or "nobody", so it gets its own branch rather than
|
||||
// going through who()'s id-vs-myID() check.
|
||||
function ackActorLabel() {
|
||||
if (inc.acknowledged_by_id != null) return who(inc.acknowledged_by_id, inc.acknowledged_by);
|
||||
if (inc.acknowledged_by_service_account_id != null) return inc.acknowledged_by_service_account || 'a service account';
|
||||
return null;
|
||||
}
|
||||
|
||||
function facts() {
|
||||
const rows = [];
|
||||
const add = (k, ...v) => rows.push(h('dt', { text: k }), h('dd', {}, ...v));
|
||||
@@ -132,8 +142,7 @@ function facts() {
|
||||
// take reading top to bottom to piece together.
|
||||
const elapsedTo = inc.resolved_at ? Date.parse(inc.resolved_at) : Date.now();
|
||||
const responsible = inc.assigned_to_id != null ? who(inc.assigned_to_id, inc.assigned_to)
|
||||
: inc.acknowledged_by_id != null ? who(inc.acknowledged_by_id, inc.acknowledged_by)
|
||||
: 'Unassigned';
|
||||
: ackActorLabel() || 'Unassigned';
|
||||
rows.push(h('dt', { text: 'At a glance' }), h('dd', { class: 'fact-summary' },
|
||||
h('span', { class: 'fact-chip' }, icon('clock', 'icon fact-icon'), duration(elapsedTo - Date.parse(inc.triggered_at))),
|
||||
inc.severity && badge(inc.severity, `plain ${severityClass(inc.severity)}`),
|
||||
@@ -142,7 +151,7 @@ function facts() {
|
||||
|
||||
add('Triggered', when(inc.triggered_at), h('span', { class: 'sub', text: ` · ${ago(inc.triggered_at)}` }));
|
||||
if (inc.acknowledged_at) {
|
||||
add('Acknowledged', `${who(inc.acknowledged_by_id, inc.acknowledged_by)} · ${when(inc.acknowledged_at)}`);
|
||||
add('Acknowledged', `${ackActorLabel()} · ${when(inc.acknowledged_at)}`);
|
||||
}
|
||||
add('Assigned', inc.assigned_to_id != null ? who(inc.assigned_to_id, inc.assigned_to) : 'Unassigned');
|
||||
if (inc.status !== 'resolved' && isFuture(inc.snoozed_until)) {
|
||||
@@ -206,9 +215,19 @@ function alertItem(a) {
|
||||
|
||||
// ---------- timeline ----------
|
||||
|
||||
// actorLabel renders whoever performed ev, human or service account — the
|
||||
// two are mutually exclusive (migration 015). null means the server acted:
|
||||
// ev.user_id == null no longer means that by itself, now that a service
|
||||
// account's events also leave it null.
|
||||
function actorLabel(ev, named = false) {
|
||||
if (ev.user_id != null) return named ? (ev.username || 'someone') : who(ev.user_id, ev.username);
|
||||
if (ev.service_account_id != null) return ev.service_account_name || 'a service account';
|
||||
return null;
|
||||
}
|
||||
|
||||
// named spells users out instead of "you", for text that leaves this page.
|
||||
function eventText(ev, named = false) {
|
||||
const person = ev.user_id != null ? (named ? ev.username || 'someone' : who(ev.user_id, ev.username)) : null;
|
||||
const person = actorLabel(ev, named);
|
||||
const strong = (t) => h('span', { class: 'who', text: t || 'someone' });
|
||||
const alertName = () => {
|
||||
const a = (inc.alerts || []).find((x) => x.id === ev.alert_id);
|
||||
@@ -453,9 +472,25 @@ function quickActions() {
|
||||
return div;
|
||||
}
|
||||
|
||||
// Desktop has room to show what a phone folds into the More sheet below — see
|
||||
// the .action-extra/.more-btn rules in app.css. Resolved/archived incidents
|
||||
// already say everything via primaryAction()/secondaryAction(), so there is
|
||||
// nothing extra to surface for them.
|
||||
function extraActions() {
|
||||
if (!isOpen()) return [];
|
||||
const out = [
|
||||
h('button', { class: 'btn btn-sm action-extra', type: 'button', onclick: assign }, icon('user'), 'Assign…'),
|
||||
h('button', { class: 'btn btn-sm action-extra', type: 'button', onclick: addNote }, icon('note'), 'Add note…'),
|
||||
];
|
||||
out.push(inc.status === 'acknowledged'
|
||||
? h('button', { class: 'btn btn-sm action-extra', type: 'button', onclick: unacknowledge }, icon('undo'), 'Clear ack')
|
||||
: h('button', { class: 'btn btn-sm action-extra', type: 'button', onclick: resolve }, icon('checkCircle'), 'Resolve…'));
|
||||
return out;
|
||||
}
|
||||
|
||||
function actionBar() {
|
||||
const more = h('button', { class: 'btn', type: 'button', 'aria-label': 'More actions', onclick: moreMenu }, icon('more'), 'More');
|
||||
const bar = h('div', { class: 'actionbar' }, primaryAction(), secondaryAction(), more);
|
||||
const more = h('button', { class: 'btn more-btn', type: 'button', 'aria-label': 'More actions', onclick: moreMenu }, icon('more'), 'More');
|
||||
const bar = h('div', { class: 'actionbar' }, primaryAction(), secondaryAction(), ...extraActions(), more);
|
||||
if (busy) for (const b of bar.querySelectorAll('button')) b.disabled = true;
|
||||
return bar;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user