Expire stale firing alerts
Release / build (amd64, darwin) (push) Failing after 12s
Release / build (arm64, darwin) (push) Failing after 11s
Release / build (arm64, linux) (push) Failing after 11s
Release / release (push) Has been skipped
Release / docker (push) Failing after 19s
Release / build (amd64, linux) (push) Failing after 12s
Release / chart (push) Failing after 9s
Release / build (amd64, darwin) (push) Failing after 12s
Release / build (arm64, darwin) (push) Failing after 11s
Release / build (arm64, linux) (push) Failing after 11s
Release / release (push) Has been skipped
Release / docker (push) Failing after 19s
Release / build (amd64, linux) (push) Failing after 12s
Release / chart (push) Failing after 9s
A resolved webhook was the only path out of the firing state, so a
notification that was dropped, silenced, or lost to a restart pinned an
alert as firing forever — Prometheus showed it resolved while
terdut-server kept listing it. The archiver only ever touched resolved
alerts, and both the list and stats queries compared status with plain
equality, so a stale row was indistinguishable from a live one.
A sweeper pass now resolves firing alerts on either of two signals: the
ends_at watermark Alertmanager sets on outgoing firing notifications has
passed (plus a grace period for clock skew), or no webhook has refreshed
the alert within TERDUT_STALE_AFTER (default 6h, above Alertmanager's 4h
repeat_interval). Such alerts get resolution_source = 'expiry',
distinguishing them from a real 'alertmanager' resolve.
Two related webhook bugs fixed alongside:
- The upsert had no ordering guard, so a retried firing notification
arriving after the resolved one resurrected the alert. Payloads for
an older alert instance are now discarded: a stale retry carries the
same startsAt, a genuine re-fire a newer one.
- archived_at was never cleared on re-fire, leaving a re-fired alert
archived and invisible in the default list.
Stats now exclude archived alerts to match the default list view; this
lowers historical firing/resolved totals.
The chart exposes both sweeper durations via sweeper.staleAfter and
sweeper.archiveAfter.
This commit is contained in:
@@ -8,6 +8,13 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// Values for alerts.resolution_source, recording why an alert left the firing
|
||||
// state: a real Alertmanager notification, or inference by the sweeper.
|
||||
const (
|
||||
resolutionAlertmanager = "alertmanager"
|
||||
resolutionExpiry = "expiry"
|
||||
)
|
||||
|
||||
// amPayload mirrors the Alertmanager webhook v4 payload.
|
||||
type amPayload struct {
|
||||
Version string `json:"version"`
|
||||
@@ -39,28 +46,51 @@ func handleAlertmanagerWebhook(db *sql.DB) http.HandlerFunc {
|
||||
labelsJSON, _ := json.Marshal(a.Labels)
|
||||
annotationsJSON, _ := json.Marshal(a.Annotations)
|
||||
|
||||
// Alertmanager uses zero time ("0001-01-01T00:00:00Z") to mean "still firing".
|
||||
// Zero time ("0001-01-01T00:00:00Z") means "no end known" — that is the
|
||||
// convention of Alertmanager's ingest API. Outgoing notifications
|
||||
// normally carry a real future endsAt instead, which is the watermark
|
||||
// the sweeper uses to expire alerts that stop being refreshed.
|
||||
var endsAtUnix *int64
|
||||
if a.EndsAt.Year() > 1 {
|
||||
t := a.EndsAt.Unix()
|
||||
endsAtUnix = &t
|
||||
}
|
||||
|
||||
var resolutionSource *string
|
||||
if a.Status == "resolved" {
|
||||
s := resolutionAlertmanager
|
||||
resolutionSource = &s
|
||||
}
|
||||
|
||||
// The WHERE clause discards payloads that describe an alert instance
|
||||
// older than the stored one. Alertmanager retries failed notifications,
|
||||
// so a stale firing retry can arrive after the resolved one; it carries
|
||||
// the same startsAt, whereas a genuine re-fire carries a newer one.
|
||||
// Within a single instance, resolution is terminal.
|
||||
_, err := db.ExecContext(r.Context(), `
|
||||
INSERT INTO alerts
|
||||
(fingerprint, name, status, labels, annotations, starts_at, ends_at, generator_url, received_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
(fingerprint, name, status, labels, annotations, starts_at, ends_at,
|
||||
generator_url, received_at, resolution_source)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(fingerprint) DO UPDATE SET
|
||||
status = excluded.status,
|
||||
labels = excluded.labels,
|
||||
annotations = excluded.annotations,
|
||||
ends_at = excluded.ends_at,
|
||||
generator_url = excluded.generator_url,
|
||||
received_at = excluded.received_at`,
|
||||
status = excluded.status,
|
||||
labels = excluded.labels,
|
||||
annotations = excluded.annotations,
|
||||
starts_at = excluded.starts_at,
|
||||
ends_at = excluded.ends_at,
|
||||
generator_url = excluded.generator_url,
|
||||
received_at = excluded.received_at,
|
||||
resolution_source = excluded.resolution_source,
|
||||
-- A re-fire makes the alert current again, so it leaves the archive.
|
||||
archived_at = CASE WHEN excluded.status = 'firing'
|
||||
THEN NULL ELSE alerts.archived_at END
|
||||
WHERE excluded.starts_at > alerts.starts_at
|
||||
OR (excluded.starts_at = alerts.starts_at
|
||||
AND NOT (alerts.status = 'resolved' AND excluded.status = 'firing'))`,
|
||||
a.Fingerprint, name, a.Status,
|
||||
string(labelsJSON), string(annotationsJSON),
|
||||
a.StartsAt.Unix(), endsAtUnix,
|
||||
a.GeneratorURL, now,
|
||||
a.GeneratorURL, now, resolutionSource,
|
||||
)
|
||||
if err != nil {
|
||||
log.Printf("upsert alert %s: %v", a.Fingerprint, err)
|
||||
|
||||
Reference in New Issue
Block a user