dc39e3a5d3
First step of #1, and it goes first for one reason: #4 adds a team_id to nearly every table, and doing that twice -- once for SQLite, once for Postgres -- is work nobody gets paid for. The teams migrations now only have to be written against one database. The ten SQLite migrations are replaced by a single Postgres baseline rather than ported one by one. They were incremental in a way that has no value on a fresh install: 004 adds columns 008 drops again, and 008's backfill rewrites data a Postgres database never had. The history stays in git; the schema they add up to is now 001_baseline.sql. Timestamps stay BIGINT unix seconds and are NOT converted to timestamptz. Everything in Go already speaks epochs, so converting would have been a second, larger change riding along inside this one. It is worth doing on its own. The JSON columns did move to jsonb, because #4 will want to filter and index on labels. Most of the port is mechanical -- 170 placeholders from ? to $1 -- but four things needed more than a search and replace: * Dynamically built WHERE clauses cannot keep their numbering straight by hand, so they hand out placeholders through sqlArgs instead. A filter can now be added or reordered without renumbering anything. * SUM(resolved_at IS NULL) was SQLite counting a boolean as 0 or 1. Postgres has no sum(boolean), and this was breaking every dead man's switch -- silently, since the sweeper only logs. Now COUNT(*) FILTER. * unixepoch() became FLOOR(EXTRACT(EPOCH FROM now()))::bigint. The FLOOR is load-bearing: a bare cast rounds half up, so a row written at .6 of a second claimed a timestamp a second in the future and disagreed with the time.Now().Unix() the Go side stamps. * The unique-violation check matched SQLite's error text. It matches SQLSTATE 23505 now, so a renamed constraint cannot turn a 409 back into a 500. Tests need a real Postgres, because there is no in-memory Postgres the way there was an in-memory SQLite. Each test gets its own schema on a shared server -- cheaper than a database each, and still isolated. TERDUT_TEST_DSN says where it is; `make test-db` starts one locally and ci.yaml runs one as a service container. An unset DSN fails the suite rather than skipping it: a run that quietly tests nothing is worse than one that does not run. TestMigration_BackfillCarriesAckAndComments is deleted along with the migrations it replayed. What it protected -- an upgrade not losing acknowledgements and comments -- now belongs to scripts/sqlite-to-postgres.go, which is build-tagged so the SQLite driver stays out of the server binary. Both are meant to be deleted once this install has migrated. The chart loses the PVC, the data volume and the python backup sidecar, and requires database.dsnSecret.name: it provisions no database and cannot guess where the credentials live, so a render without it is meant to fail. Backups move to where Postgres actually runs. The other half of that -- the postgresql CR, the k8up pg_dump annotation and the network policy -- is a change to the wrapper chart in Ryuvia/charts and is not in here. Verified rather than assumed: the gate is green with -race against Postgres 17, govulncheck and gitleaks are clean, and the migration script was run end to end against a SQLite database built at the old schema and seeded in every table. Ids survive, so incidents keep their numbers and every foreign key still points where it did; the identity sequences are moved past the copied ids, and a webhook after the migration opened incident 12 rather than colliding at 1.
240 lines
7.6 KiB
Go
240 lines
7.6 KiB
Go
package api
|
|
|
|
import (
|
|
"context"
|
|
"database/sql"
|
|
"log"
|
|
|
|
"time"
|
|
)
|
|
|
|
const (
|
|
// sweepInterval is how often the background sweeper runs.
|
|
sweepInterval = 15 * time.Minute
|
|
|
|
// expiryGrace absorbs clock skew and notification latency before an alert
|
|
// whose ends_at watermark has passed is treated as stale.
|
|
expiryGrace = 5 * time.Minute
|
|
)
|
|
|
|
// StartArchiver runs the alert sweeper until ctx is cancelled, starting with an
|
|
// immediate pass so a restart reconciles state right away.
|
|
func StartArchiver(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, deadman DeadmanConfig, notify NotifyConfig) {
|
|
ticker := time.NewTicker(sweepInterval)
|
|
defer ticker.Stop()
|
|
|
|
Sweep(ctx, db, archiveAfter, staleAfter, deadman, notify)
|
|
for {
|
|
select {
|
|
case <-ticker.C:
|
|
Sweep(ctx, db, archiveAfter, staleAfter, deadman, notify)
|
|
case <-ctx.Done():
|
|
return
|
|
}
|
|
}
|
|
}
|
|
|
|
// Sweep runs a single pass, in dependency order: reconcile the dead man's
|
|
// switches, expire stale firing alerts, close the incidents that leaves with
|
|
// nothing firing, then archive whatever has been settled long enough. Running
|
|
// them in one pass means an alert can go stale and its incident can close and
|
|
// archive without waiting three ticks.
|
|
//
|
|
// The switches go first because they hand expireStale the alerts it must not
|
|
// touch: a heartbeat answers to its own, much tighter, timeout, and the generic
|
|
// staleness rules would otherwise resolve it as 'expiry' long before that.
|
|
// Exported so tests can drive a pass without waiting on the ticker.
|
|
func Sweep(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, deadman DeadmanConfig, notify NotifyConfig) {
|
|
heartbeats := sweepDeadman(ctx, db, deadman, notify)
|
|
expireStale(ctx, db, staleAfter, heartbeats)
|
|
resolveSettledIncidents(ctx, db)
|
|
archiveResolved(ctx, db, archiveAfter)
|
|
archiveResolvedIncidents(ctx, db, archiveAfter)
|
|
purgeAckTokens(ctx, db)
|
|
purgeSessions(ctx, db)
|
|
}
|
|
|
|
// expireStale resolves firing alerts that Alertmanager has stopped refreshing.
|
|
//
|
|
// A resolved webhook is otherwise the only way out of the firing state, so a
|
|
// notification that is dropped, silenced, or lost to a restart would pin the
|
|
// alert as firing forever. Two independent signals mark an alert stale:
|
|
//
|
|
// - ends_at, the "valid until" watermark Alertmanager sets on outgoing firing
|
|
// notifications, has passed (plus expiryGrace for clock skew). Absent on
|
|
// rows whose payload carried no ends_at, hence the second signal.
|
|
// - received_at is older than staleAfter. Alertmanager re-sends firing
|
|
// notifications every repeat_interval, making received_at a liveness
|
|
// heartbeat — provided staleAfter exceeds that interval.
|
|
//
|
|
// Alerts in skip are left alone: they are dead man's switch heartbeats, whose
|
|
// liveness sweepDeadman has already judged against a timeout of its own.
|
|
//
|
|
// The matching rows are collected before the update rather than updated in bulk,
|
|
// because each one owes its incident a timeline entry.
|
|
func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip map[int64]bool) {
|
|
now := time.Now()
|
|
|
|
found, err := staleAlertIDs(ctx, db, now, staleAfter)
|
|
if err != nil {
|
|
log.Printf("sweeper: find stale: %v", err)
|
|
return
|
|
}
|
|
|
|
ids := make([]int64, 0, len(found))
|
|
for _, id := range found {
|
|
if !skip[id] {
|
|
ids = append(ids, id)
|
|
}
|
|
}
|
|
if len(ids) == 0 {
|
|
return
|
|
}
|
|
|
|
args := &sqlArgs{}
|
|
source := args.add(resolutionExpiry)
|
|
idList := make([]any, len(ids))
|
|
for i, id := range ids {
|
|
idList[i] = id
|
|
}
|
|
if _, err := db.ExecContext(ctx, `
|
|
UPDATE alerts
|
|
SET status = 'resolved',
|
|
resolution_source = `+source+`,
|
|
ends_at = COALESCE(ends_at, `+nowEpoch+`)
|
|
WHERE id IN (`+args.addList(idList)+`)`, args.all()...); err != nil {
|
|
log.Printf("sweeper: expire stale: %v", err)
|
|
return
|
|
}
|
|
log.Printf("sweeper: expired %d stale firing alert(s)", len(ids))
|
|
|
|
for _, id := range ids {
|
|
incidentID, err := openIncidentForAlert(ctx, db, id)
|
|
if err != nil {
|
|
log.Printf("sweeper: incident for alert %d: %v", id, err)
|
|
continue
|
|
}
|
|
if incidentID == 0 {
|
|
continue
|
|
}
|
|
alertID := id
|
|
if err := logEvent(ctx, db, incidentID, evAlertResolved, nil, &alertID, nil); err != nil {
|
|
log.Printf("sweeper: log expiry event: %v", err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// staleAlertIDs reads the ids in one go and closes the cursor before the caller
|
|
// writes. Under SQLite's single connection an open read would have blocked the
|
|
// update outright; with a pool it is no longer a deadlock, but reading the set
|
|
// first still keeps the write off a cursor the same transaction is walking.
|
|
func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) {
|
|
rows, err := db.QueryContext(ctx, `
|
|
SELECT id FROM alerts
|
|
WHERE status = 'firing'
|
|
AND archived_at IS NULL
|
|
AND ((ends_at IS NOT NULL AND ends_at < $1) OR received_at < $2)`,
|
|
now.Add(-expiryGrace).Unix(), now.Add(-staleAfter).Unix())
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer rows.Close()
|
|
|
|
var ids []int64
|
|
for rows.Next() {
|
|
var id int64
|
|
if err := rows.Scan(&id); err != nil {
|
|
return nil, err
|
|
}
|
|
ids = append(ids, id)
|
|
}
|
|
return ids, rows.Err()
|
|
}
|
|
|
|
// resolveSettledIncidents closes incidents whose alerts have all stopped firing.
|
|
// This is the cascade from alerts up to the work item, and it is what turns an
|
|
// expiry into a closed incident rather than one that sits open forever.
|
|
func resolveSettledIncidents(ctx context.Context, db *sql.DB) {
|
|
ids, err := settledIncidentIDs(ctx, db)
|
|
if err != nil {
|
|
log.Printf("sweeper: find settled incidents: %v", err)
|
|
return
|
|
}
|
|
|
|
resolved := 0
|
|
for _, id := range ids {
|
|
ok, err := resolveIfSettled(ctx, db, id)
|
|
if err != nil {
|
|
log.Printf("sweeper: resolve incident %d: %v", id, err)
|
|
continue
|
|
}
|
|
if ok {
|
|
resolved++
|
|
}
|
|
}
|
|
if resolved > 0 {
|
|
log.Printf("sweeper: resolved %d settled incident(s)", resolved)
|
|
}
|
|
}
|
|
|
|
func settledIncidentIDs(ctx context.Context, db *sql.DB) ([]int64, error) {
|
|
rows, err := db.QueryContext(ctx, `
|
|
SELECT i.id
|
|
FROM incidents i
|
|
WHERE i.resolved_at IS NULL
|
|
AND EXISTS (SELECT 1 FROM incident_alerts ia WHERE ia.incident_id = i.id)
|
|
AND NOT EXISTS (SELECT 1
|
|
FROM incident_alerts ia
|
|
JOIN alerts a ON a.id = ia.alert_id
|
|
WHERE ia.incident_id = i.id
|
|
AND a.status = 'firing')`)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer rows.Close()
|
|
|
|
var ids []int64
|
|
for rows.Next() {
|
|
var id int64
|
|
if err := rows.Scan(&id); err != nil {
|
|
return nil, err
|
|
}
|
|
ids = append(ids, id)
|
|
}
|
|
return ids, rows.Err()
|
|
}
|
|
|
|
// archiveResolved hides resolved alerts that have been settled for archiveAfter.
|
|
func archiveResolved(ctx context.Context, db *sql.DB, archiveAfter time.Duration) {
|
|
cutoff := time.Now().Add(-archiveAfter).Unix()
|
|
res, err := db.ExecContext(ctx,
|
|
`UPDATE alerts SET archived_at = `+nowEpoch+`
|
|
WHERE status = 'resolved'
|
|
AND archived_at IS NULL
|
|
AND COALESCE(ends_at, received_at) < $1`, cutoff)
|
|
if err != nil {
|
|
log.Printf("archiver: %v", err)
|
|
return
|
|
}
|
|
if n, _ := res.RowsAffected(); n > 0 {
|
|
log.Printf("archiver: archived %d resolved alert(s)", n)
|
|
}
|
|
}
|
|
|
|
// archiveResolvedIncidents does the same for the work items, on the same clock.
|
|
func archiveResolvedIncidents(ctx context.Context, db *sql.DB, archiveAfter time.Duration) {
|
|
cutoff := time.Now().Add(-archiveAfter).Unix()
|
|
res, err := db.ExecContext(ctx,
|
|
`UPDATE incidents SET archived_at = `+nowEpoch+`
|
|
WHERE resolved_at IS NOT NULL
|
|
AND archived_at IS NULL
|
|
AND resolved_at < $1`, cutoff)
|
|
if err != nil {
|
|
log.Printf("archiver: incidents: %v", err)
|
|
return
|
|
}
|
|
if n, _ := res.RowsAffected(); n > 0 {
|
|
log.Printf("archiver: archived %d resolved incident(s)", n)
|
|
}
|
|
}
|