Expire stale firing alerts
Release / build (amd64, darwin) (push) Failing after 12s
Release / build (arm64, darwin) (push) Failing after 11s
Release / build (arm64, linux) (push) Failing after 11s
Release / release (push) Has been skipped
Release / docker (push) Failing after 19s
Release / build (amd64, linux) (push) Failing after 12s
Release / chart (push) Failing after 9s
Release / build (amd64, darwin) (push) Failing after 12s
Release / build (arm64, darwin) (push) Failing after 11s
Release / build (arm64, linux) (push) Failing after 11s
Release / release (push) Has been skipped
Release / docker (push) Failing after 19s
Release / build (amd64, linux) (push) Failing after 12s
Release / chart (push) Failing after 9s
A resolved webhook was the only path out of the firing state, so a
notification that was dropped, silenced, or lost to a restart pinned an
alert as firing forever — Prometheus showed it resolved while
terdut-server kept listing it. The archiver only ever touched resolved
alerts, and both the list and stats queries compared status with plain
equality, so a stale row was indistinguishable from a live one.
A sweeper pass now resolves firing alerts on either of two signals: the
ends_at watermark Alertmanager sets on outgoing firing notifications has
passed (plus a grace period for clock skew), or no webhook has refreshed
the alert within TERDUT_STALE_AFTER (default 6h, above Alertmanager's 4h
repeat_interval). Such alerts get resolution_source = 'expiry',
distinguishing them from a real 'alertmanager' resolve.
Two related webhook bugs fixed alongside:
- The upsert had no ordering guard, so a retried firing notification
arriving after the resolved one resurrected the alert. Payloads for
an older alert instance are now discarded: a stale retry carries the
same startsAt, a genuine re-fire a newer one.
- archived_at was never cleared on re-fire, leaving a re-fired alert
archived and invisible in the default list.
Stats now exclude archived alerts to match the default list view; this
lowers historical firing/resolved totals.
The chart exposes both sweeper durations via sweeper.staleAfter and
sweeper.archiveAfter.
This commit is contained in:
+211
-2
@@ -2,21 +2,26 @@ package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/yeniklas/terdut-server/internal/api"
|
||||
"github.com/yeniklas/terdut-server/internal/db"
|
||||
)
|
||||
|
||||
// ts wraps httptest.Server with a pre-bootstrapped API key.
|
||||
// ts wraps httptest.Server with a pre-bootstrapped API key. db is exposed so
|
||||
// tests can age rows directly — the sweeper's inputs are wall-clock timestamps.
|
||||
type ts struct {
|
||||
*httptest.Server
|
||||
key string
|
||||
db *sql.DB
|
||||
}
|
||||
|
||||
func newTS(t *testing.T) *ts {
|
||||
@@ -44,7 +49,27 @@ func newTS(t *testing.T) *ts {
|
||||
json.NewDecoder(resp.Body).Decode(&result)
|
||||
key := result["api_key"].(map[string]any)["key"].(string)
|
||||
|
||||
return &ts{Server: srv, key: key}
|
||||
return &ts{Server: srv, key: key, db: database}
|
||||
}
|
||||
|
||||
// exec runs a statement against the test database.
|
||||
func (s *ts) exec(t *testing.T, query string, args ...any) {
|
||||
t.Helper()
|
||||
if _, err := s.db.Exec(query, args...); err != nil {
|
||||
t.Fatalf("exec %q: %v", query, err)
|
||||
}
|
||||
}
|
||||
|
||||
// alertRow reads the sweeper-relevant columns of one alert straight from the DB.
|
||||
func (s *ts) alertRow(t *testing.T, fingerprint string) (status string, source *string, archivedAt *int64) {
|
||||
t.Helper()
|
||||
err := s.db.QueryRow(
|
||||
"SELECT status, resolution_source, archived_at FROM alerts WHERE fingerprint = ?",
|
||||
fingerprint).Scan(&status, &source, &archivedAt)
|
||||
if err != nil {
|
||||
t.Fatalf("read alert %s: %v", fingerprint, err)
|
||||
}
|
||||
return status, source, archivedAt
|
||||
}
|
||||
|
||||
// req sends an authenticated request, optionally with a JSON body.
|
||||
@@ -417,6 +442,190 @@ func TestArchive_RoundTrip(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Stale-alert expiry
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// noArchive is long enough that archiving never interferes with expiry tests.
|
||||
const noArchive = 365 * 24 * time.Hour
|
||||
|
||||
// zeroTime is Alertmanager's "no end known" sentinel, which stores ends_at NULL.
|
||||
const zeroTime = "0001-01-01T00:00:00Z"
|
||||
|
||||
// postAlert sends a single-alert webhook.
|
||||
func postAlert(t *testing.T, s *ts, fingerprint, status, startsAt, endsAt string) {
|
||||
t.Helper()
|
||||
postWebhook(t, s, []map[string]any{{
|
||||
"status": status,
|
||||
"labels": map[string]string{"alertname": "Stale"},
|
||||
"annotations": map[string]string{},
|
||||
"startsAt": startsAt,
|
||||
"endsAt": endsAt,
|
||||
"generatorURL": "",
|
||||
"fingerprint": fingerprint,
|
||||
}})
|
||||
}
|
||||
|
||||
func sweep(t *testing.T, s *ts, staleAfter time.Duration) {
|
||||
t.Helper()
|
||||
api.Sweep(context.Background(), s.db, noArchive, staleAfter)
|
||||
}
|
||||
|
||||
// A firing alert Alertmanager stopped refreshing is resolved via the
|
||||
// received_at heartbeat, even with no ends_at watermark to go on.
|
||||
func TestExpiry_StaleFiringAlert(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postAlert(t, s, "stale1", "firing", time.Now().Add(-24*time.Hour).Format(time.RFC3339), zeroTime)
|
||||
|
||||
// Age the last-seen timestamp past the staleness window.
|
||||
s.exec(t, "UPDATE alerts SET received_at = ? WHERE fingerprint = 'stale1'",
|
||||
time.Now().Add(-10*time.Hour).Unix())
|
||||
|
||||
sweep(t, s, 6*time.Hour)
|
||||
|
||||
status, source, _ := s.alertRow(t, "stale1")
|
||||
if status != "resolved" {
|
||||
t.Errorf("expected status resolved, got %q", status)
|
||||
}
|
||||
if source == nil || *source != "expiry" {
|
||||
t.Errorf("expected resolution_source=expiry, got %v", source)
|
||||
}
|
||||
}
|
||||
|
||||
// A fresh webhook whose ends_at watermark has already passed is expired without
|
||||
// waiting out the full staleness window.
|
||||
func TestExpiry_PastEndsAt(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postAlert(t, s, "stale2", "firing",
|
||||
time.Now().Add(-2*time.Hour).Format(time.RFC3339),
|
||||
time.Now().Add(-30*time.Minute).Format(time.RFC3339))
|
||||
|
||||
sweep(t, s, 6*time.Hour) // received_at is fresh; only ends_at can trigger
|
||||
|
||||
status, source, _ := s.alertRow(t, "stale2")
|
||||
if status != "resolved" {
|
||||
t.Errorf("expected status resolved, got %q", status)
|
||||
}
|
||||
if source == nil || *source != "expiry" {
|
||||
t.Errorf("expected resolution_source=expiry, got %v", source)
|
||||
}
|
||||
}
|
||||
|
||||
// The regression that matters most: a genuinely firing alert must survive a
|
||||
// sweep untouched.
|
||||
func TestExpiry_LeavesFreshAlertsAlone(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postAlert(t, s, "fresh1", "firing",
|
||||
time.Now().Add(-10*time.Minute).Format(time.RFC3339),
|
||||
time.Now().Add(1*time.Hour).Format(time.RFC3339))
|
||||
|
||||
sweep(t, s, 6*time.Hour)
|
||||
|
||||
status, source, _ := s.alertRow(t, "fresh1")
|
||||
if status != "firing" {
|
||||
t.Errorf("expected fresh alert to stay firing, got %q", status)
|
||||
}
|
||||
if source != nil {
|
||||
t.Errorf("expected no resolution_source, got %q", *source)
|
||||
}
|
||||
}
|
||||
|
||||
// An ends_at only just past must not trip expiry — that grace absorbs clock skew.
|
||||
func TestExpiry_RespectsGraceOnEndsAt(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postAlert(t, s, "grace1", "firing",
|
||||
time.Now().Add(-time.Hour).Format(time.RFC3339),
|
||||
time.Now().Add(-1*time.Minute).Format(time.RFC3339))
|
||||
|
||||
sweep(t, s, 6*time.Hour)
|
||||
|
||||
if status, _, _ := s.alertRow(t, "grace1"); status != "firing" {
|
||||
t.Errorf("expected alert within grace period to stay firing, got %q", status)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Webhook resolution bookkeeping
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestWebhook_ResolvedSetsSource(t *testing.T) {
|
||||
s := newTS(t)
|
||||
start := time.Now().Add(-time.Hour).Format(time.RFC3339)
|
||||
postAlert(t, s, "src1", "firing", start, zeroTime)
|
||||
|
||||
if _, source, _ := s.alertRow(t, "src1"); source != nil {
|
||||
t.Errorf("expected firing alert to have no resolution_source, got %q", *source)
|
||||
}
|
||||
|
||||
postAlert(t, s, "src1", "resolved", start, time.Now().Format(time.RFC3339))
|
||||
|
||||
status, source, _ := s.alertRow(t, "src1")
|
||||
if status != "resolved" {
|
||||
t.Errorf("expected status resolved, got %q", status)
|
||||
}
|
||||
if source == nil || *source != "alertmanager" {
|
||||
t.Errorf("expected resolution_source=alertmanager, got %v", source)
|
||||
}
|
||||
}
|
||||
|
||||
// A re-fire under the same fingerprint must leave the archive and clear the
|
||||
// stale expiry marker, otherwise the alert stays invisible in the default list.
|
||||
func TestWebhook_RefireUnarchivesAndClearsSource(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postAlert(t, s, "refire1", "firing", time.Now().Add(-24*time.Hour).Format(time.RFC3339), zeroTime)
|
||||
|
||||
// Expire it, then archive it.
|
||||
s.exec(t, "UPDATE alerts SET received_at = ? WHERE fingerprint = 'refire1'",
|
||||
time.Now().Add(-10*time.Hour).Unix())
|
||||
sweep(t, s, 6*time.Hour)
|
||||
s.exec(t, "UPDATE alerts SET archived_at = unixepoch() WHERE fingerprint = 'refire1'")
|
||||
|
||||
var alerts []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
|
||||
if len(alerts) != 0 {
|
||||
t.Fatalf("expected archived alert to be hidden, got %d", len(alerts))
|
||||
}
|
||||
|
||||
// Fires again: a new alert instance, so a newer startsAt.
|
||||
postAlert(t, s, "refire1", "firing", time.Now().Format(time.RFC3339), zeroTime)
|
||||
|
||||
status, source, archivedAt := s.alertRow(t, "refire1")
|
||||
if status != "firing" {
|
||||
t.Errorf("expected status firing after re-fire, got %q", status)
|
||||
}
|
||||
if source != nil {
|
||||
t.Errorf("expected resolution_source cleared on re-fire, got %q", *source)
|
||||
}
|
||||
if archivedAt != nil {
|
||||
t.Errorf("expected archived_at cleared on re-fire, got %d", *archivedAt)
|
||||
}
|
||||
|
||||
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
|
||||
if len(alerts) != 1 {
|
||||
t.Errorf("expected re-fired alert back in default list, got %d", len(alerts))
|
||||
}
|
||||
}
|
||||
|
||||
// Alertmanager retries failed notifications, so a firing payload for an
|
||||
// already-resolved instance can arrive late. It must not resurrect the alert.
|
||||
func TestWebhook_IgnoresOutOfOrderRetry(t *testing.T) {
|
||||
s := newTS(t)
|
||||
start := time.Now().Add(-time.Hour).Format(time.RFC3339)
|
||||
end := time.Now().Format(time.RFC3339)
|
||||
|
||||
postAlert(t, s, "ooo1", "firing", start, zeroTime)
|
||||
postAlert(t, s, "ooo1", "resolved", start, end)
|
||||
postAlert(t, s, "ooo1", "firing", start, zeroTime) // stale retry, same instance
|
||||
|
||||
status, source, _ := s.alertRow(t, "ooo1")
|
||||
if status != "resolved" {
|
||||
t.Errorf("expected alert to stay resolved after stale retry, got %q", status)
|
||||
}
|
||||
if source == nil || *source != "alertmanager" {
|
||||
t.Errorf("expected resolution_source=alertmanager, got %v", source)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStats_ByDayReturnsSevenSlots(t *testing.T) {
|
||||
s := newTS(t)
|
||||
resp := s.req(t, http.MethodGet, "/api/stats/alerts/by-day", nil)
|
||||
|
||||
Reference in New Issue
Block a user