Expire stale firing alerts
Release / build (amd64, darwin) (push) Failing after 12s
Release / build (arm64, darwin) (push) Failing after 11s
Release / build (arm64, linux) (push) Failing after 11s
Release / release (push) Has been skipped
Release / docker (push) Failing after 19s
Release / build (amd64, linux) (push) Failing after 12s
Release / chart (push) Failing after 9s

A resolved webhook was the only path out of the firing state, so a
notification that was dropped, silenced, or lost to a restart pinned an
alert as firing forever — Prometheus showed it resolved while
terdut-server kept listing it. The archiver only ever touched resolved
alerts, and both the list and stats queries compared status with plain
equality, so a stale row was indistinguishable from a live one.

A sweeper pass now resolves firing alerts on either of two signals: the
ends_at watermark Alertmanager sets on outgoing firing notifications has
passed (plus a grace period for clock skew), or no webhook has refreshed
the alert within TERDUT_STALE_AFTER (default 6h, above Alertmanager's 4h
repeat_interval). Such alerts get resolution_source = 'expiry',
distinguishing them from a real 'alertmanager' resolve.

Two related webhook bugs fixed alongside:

  - The upsert had no ordering guard, so a retried firing notification
    arriving after the resolved one resurrected the alert. Payloads for
    an older alert instance are now discarded: a stale retry carries the
    same startsAt, a genuine re-fire a newer one.
  - archived_at was never cleared on re-fire, leaving a re-fired alert
    archived and invisible in the default list.

Stats now exclude archived alerts to match the default list view; this
lowers historical firing/resolved totals.

The chart exposes both sweeper durations via sweeper.staleAfter and
sweeper.archiveAfter.
This commit is contained in:
Niklas Ye
2026-07-28 11:49:39 +02:00
parent debc4bf78c
commit 42e846f876
12 changed files with 391 additions and 42 deletions
+211 -2
View File
@@ -2,21 +2,26 @@ package api_test
import (
"bytes"
"context"
"database/sql"
"encoding/json"
"fmt"
"io"
"net/http"
"net/http/httptest"
"testing"
"time"
"github.com/yeniklas/terdut-server/internal/api"
"github.com/yeniklas/terdut-server/internal/db"
)
// ts wraps httptest.Server with a pre-bootstrapped API key.
// ts wraps httptest.Server with a pre-bootstrapped API key. db is exposed so
// tests can age rows directly — the sweeper's inputs are wall-clock timestamps.
type ts struct {
*httptest.Server
key string
db *sql.DB
}
func newTS(t *testing.T) *ts {
@@ -44,7 +49,27 @@ func newTS(t *testing.T) *ts {
json.NewDecoder(resp.Body).Decode(&result)
key := result["api_key"].(map[string]any)["key"].(string)
return &ts{Server: srv, key: key}
return &ts{Server: srv, key: key, db: database}
}
// exec runs a statement against the test database.
func (s *ts) exec(t *testing.T, query string, args ...any) {
t.Helper()
if _, err := s.db.Exec(query, args...); err != nil {
t.Fatalf("exec %q: %v", query, err)
}
}
// alertRow reads the sweeper-relevant columns of one alert straight from the DB.
func (s *ts) alertRow(t *testing.T, fingerprint string) (status string, source *string, archivedAt *int64) {
t.Helper()
err := s.db.QueryRow(
"SELECT status, resolution_source, archived_at FROM alerts WHERE fingerprint = ?",
fingerprint).Scan(&status, &source, &archivedAt)
if err != nil {
t.Fatalf("read alert %s: %v", fingerprint, err)
}
return status, source, archivedAt
}
// req sends an authenticated request, optionally with a JSON body.
@@ -417,6 +442,190 @@ func TestArchive_RoundTrip(t *testing.T) {
}
}
// ---------------------------------------------------------------------------
// Stale-alert expiry
// ---------------------------------------------------------------------------
// noArchive is long enough that archiving never interferes with expiry tests.
const noArchive = 365 * 24 * time.Hour
// zeroTime is Alertmanager's "no end known" sentinel, which stores ends_at NULL.
const zeroTime = "0001-01-01T00:00:00Z"
// postAlert sends a single-alert webhook.
func postAlert(t *testing.T, s *ts, fingerprint, status, startsAt, endsAt string) {
t.Helper()
postWebhook(t, s, []map[string]any{{
"status": status,
"labels": map[string]string{"alertname": "Stale"},
"annotations": map[string]string{},
"startsAt": startsAt,
"endsAt": endsAt,
"generatorURL": "",
"fingerprint": fingerprint,
}})
}
func sweep(t *testing.T, s *ts, staleAfter time.Duration) {
t.Helper()
api.Sweep(context.Background(), s.db, noArchive, staleAfter)
}
// A firing alert Alertmanager stopped refreshing is resolved via the
// received_at heartbeat, even with no ends_at watermark to go on.
func TestExpiry_StaleFiringAlert(t *testing.T) {
s := newTS(t)
postAlert(t, s, "stale1", "firing", time.Now().Add(-24*time.Hour).Format(time.RFC3339), zeroTime)
// Age the last-seen timestamp past the staleness window.
s.exec(t, "UPDATE alerts SET received_at = ? WHERE fingerprint = 'stale1'",
time.Now().Add(-10*time.Hour).Unix())
sweep(t, s, 6*time.Hour)
status, source, _ := s.alertRow(t, "stale1")
if status != "resolved" {
t.Errorf("expected status resolved, got %q", status)
}
if source == nil || *source != "expiry" {
t.Errorf("expected resolution_source=expiry, got %v", source)
}
}
// A fresh webhook whose ends_at watermark has already passed is expired without
// waiting out the full staleness window.
func TestExpiry_PastEndsAt(t *testing.T) {
s := newTS(t)
postAlert(t, s, "stale2", "firing",
time.Now().Add(-2*time.Hour).Format(time.RFC3339),
time.Now().Add(-30*time.Minute).Format(time.RFC3339))
sweep(t, s, 6*time.Hour) // received_at is fresh; only ends_at can trigger
status, source, _ := s.alertRow(t, "stale2")
if status != "resolved" {
t.Errorf("expected status resolved, got %q", status)
}
if source == nil || *source != "expiry" {
t.Errorf("expected resolution_source=expiry, got %v", source)
}
}
// The regression that matters most: a genuinely firing alert must survive a
// sweep untouched.
func TestExpiry_LeavesFreshAlertsAlone(t *testing.T) {
s := newTS(t)
postAlert(t, s, "fresh1", "firing",
time.Now().Add(-10*time.Minute).Format(time.RFC3339),
time.Now().Add(1*time.Hour).Format(time.RFC3339))
sweep(t, s, 6*time.Hour)
status, source, _ := s.alertRow(t, "fresh1")
if status != "firing" {
t.Errorf("expected fresh alert to stay firing, got %q", status)
}
if source != nil {
t.Errorf("expected no resolution_source, got %q", *source)
}
}
// An ends_at only just past must not trip expiry — that grace absorbs clock skew.
func TestExpiry_RespectsGraceOnEndsAt(t *testing.T) {
s := newTS(t)
postAlert(t, s, "grace1", "firing",
time.Now().Add(-time.Hour).Format(time.RFC3339),
time.Now().Add(-1*time.Minute).Format(time.RFC3339))
sweep(t, s, 6*time.Hour)
if status, _, _ := s.alertRow(t, "grace1"); status != "firing" {
t.Errorf("expected alert within grace period to stay firing, got %q", status)
}
}
// ---------------------------------------------------------------------------
// Webhook resolution bookkeeping
// ---------------------------------------------------------------------------
func TestWebhook_ResolvedSetsSource(t *testing.T) {
s := newTS(t)
start := time.Now().Add(-time.Hour).Format(time.RFC3339)
postAlert(t, s, "src1", "firing", start, zeroTime)
if _, source, _ := s.alertRow(t, "src1"); source != nil {
t.Errorf("expected firing alert to have no resolution_source, got %q", *source)
}
postAlert(t, s, "src1", "resolved", start, time.Now().Format(time.RFC3339))
status, source, _ := s.alertRow(t, "src1")
if status != "resolved" {
t.Errorf("expected status resolved, got %q", status)
}
if source == nil || *source != "alertmanager" {
t.Errorf("expected resolution_source=alertmanager, got %v", source)
}
}
// A re-fire under the same fingerprint must leave the archive and clear the
// stale expiry marker, otherwise the alert stays invisible in the default list.
func TestWebhook_RefireUnarchivesAndClearsSource(t *testing.T) {
s := newTS(t)
postAlert(t, s, "refire1", "firing", time.Now().Add(-24*time.Hour).Format(time.RFC3339), zeroTime)
// Expire it, then archive it.
s.exec(t, "UPDATE alerts SET received_at = ? WHERE fingerprint = 'refire1'",
time.Now().Add(-10*time.Hour).Unix())
sweep(t, s, 6*time.Hour)
s.exec(t, "UPDATE alerts SET archived_at = unixepoch() WHERE fingerprint = 'refire1'")
var alerts []map[string]any
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
if len(alerts) != 0 {
t.Fatalf("expected archived alert to be hidden, got %d", len(alerts))
}
// Fires again: a new alert instance, so a newer startsAt.
postAlert(t, s, "refire1", "firing", time.Now().Format(time.RFC3339), zeroTime)
status, source, archivedAt := s.alertRow(t, "refire1")
if status != "firing" {
t.Errorf("expected status firing after re-fire, got %q", status)
}
if source != nil {
t.Errorf("expected resolution_source cleared on re-fire, got %q", *source)
}
if archivedAt != nil {
t.Errorf("expected archived_at cleared on re-fire, got %d", *archivedAt)
}
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
if len(alerts) != 1 {
t.Errorf("expected re-fired alert back in default list, got %d", len(alerts))
}
}
// Alertmanager retries failed notifications, so a firing payload for an
// already-resolved instance can arrive late. It must not resurrect the alert.
func TestWebhook_IgnoresOutOfOrderRetry(t *testing.T) {
s := newTS(t)
start := time.Now().Add(-time.Hour).Format(time.RFC3339)
end := time.Now().Format(time.RFC3339)
postAlert(t, s, "ooo1", "firing", start, zeroTime)
postAlert(t, s, "ooo1", "resolved", start, end)
postAlert(t, s, "ooo1", "firing", start, zeroTime) // stale retry, same instance
status, source, _ := s.alertRow(t, "ooo1")
if status != "resolved" {
t.Errorf("expected alert to stay resolved after stale retry, got %q", status)
}
if source == nil || *source != "alertmanager" {
t.Errorf("expected resolution_source=alertmanager, got %v", source)
}
}
func TestStats_ByDayReturnsSevenSlots(t *testing.T) {
s := newTS(t)
resp := s.req(t, http.MethodGet, "/api/stats/alerts/by-day", nil)