Guard the archiver and notifier passes with a Postgres advisory lock
Both background loops run unconditionally on every instance with no coordination between them, which the chart's replicas: 1 + strategy: Recreate exists specifically to paper over: with more than one replica, every one of them would sweep and deliver notifications independently, and two overlapping during a rollout would both page for the same incident. Add withAdvisoryLock, which takes a Postgres advisory lock on a dedicated connection and runs a pass only if it gets the lock, otherwise skipping until the next tick. Wire StartArchiver and StartNotifier through it with their own lock keys, so Sweep and NotifySweep themselves are untouched and every existing test calling them directly keeps working unchanged. This also closes the notifier's double-delivery race in passing: two replicas can no longer both be inside deliverPending at once, since only one can hold notifierLockKey at a time. Deliberately not addressed here, and still blocking a replica count above 1: the in-memory login rate limiter, the unlocked migration runner, and the new-incident-insert race on a webhook for a brand-new groupKey. Noted in the updated chart comment. Co-authored-by: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -1,8 +1,11 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -77,3 +80,39 @@ func decodeJSON(r *http.Request, v any) error {
|
||||
func errResp(msg string) map[string]string {
|
||||
return map[string]string{"error": msg}
|
||||
}
|
||||
|
||||
// withAdvisoryLock runs fn only if it can take the named Postgres advisory lock on a
|
||||
// dedicated connection, and skips fn otherwise. This is what keeps the archiver and
|
||||
// notifier safe to run on more than one replica: whichever instance's tick gets there
|
||||
// first does the work; the rest see the lock held and simply wait for their next tick
|
||||
// instead of running the same pass concurrently.
|
||||
//
|
||||
// pg_try_advisory_lock is session-scoped, so taking and releasing it must happen on the
|
||||
// same connection, reserved via db.Conn rather than borrowed from the pool's shared
|
||||
// connections fn itself may use — and released (unlocked, then closed) before returning,
|
||||
// since a session lock otherwise outlives this call and leaks onto whatever reuses the
|
||||
// pooled connection next.
|
||||
func withAdvisoryLock(ctx context.Context, db *sql.DB, key int64, name string, fn func()) {
|
||||
conn, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
log.Printf("%s: advisory lock: acquire connection: %v", name, err)
|
||||
return
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
var locked bool
|
||||
if err := conn.QueryRowContext(ctx, "SELECT pg_try_advisory_lock($1)", key).Scan(&locked); err != nil {
|
||||
log.Printf("%s: advisory lock: %v", name, err)
|
||||
return
|
||||
}
|
||||
if !locked {
|
||||
return // another replica is already running this pass
|
||||
}
|
||||
defer func() {
|
||||
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", key); err != nil {
|
||||
log.Printf("%s: advisory unlock: %v", name, err)
|
||||
}
|
||||
}()
|
||||
|
||||
fn()
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user