42180948d1
Both background loops run unconditionally on every instance with no coordination between them, which the chart's replicas: 1 + strategy: Recreate exists specifically to paper over: with more than one replica, every one of them would sweep and deliver notifications independently, and two overlapping during a rollout would both page for the same incident. Add withAdvisoryLock, which takes a Postgres advisory lock on a dedicated connection and runs a pass only if it gets the lock, otherwise skipping until the next tick. Wire StartArchiver and StartNotifier through it with their own lock keys, so Sweep and NotifySweep themselves are untouched and every existing test calling them directly keeps working unchanged. This also closes the notifier's double-delivery race in passing: two replicas can no longer both be inside deliverPending at once, since only one can hold notifierLockKey at a time. Deliberately not addressed here, and still blocking a replica count above 1: the in-memory login rate limiter, the unlocked migration runner, and the new-incident-insert race on a webhook for a brand-new groupKey. Noted in the updated chart comment. Co-authored-by: Claude <noreply@anthropic.com>
119 lines
4.3 KiB
Go
119 lines
4.3 KiB
Go
package api
|
|
|
|
import (
|
|
"context"
|
|
"database/sql"
|
|
"encoding/json"
|
|
"errors"
|
|
"log"
|
|
"net/http"
|
|
"strconv"
|
|
"strings"
|
|
|
|
"github.com/jackc/pgerrcode"
|
|
"github.com/jackc/pgx/v5/pgconn"
|
|
)
|
|
|
|
// sqlArgs accumulates query arguments and hands back the placeholder for each.
|
|
//
|
|
// Postgres numbers its placeholders, so a dynamically assembled WHERE clause has
|
|
// to keep its $1, $2, … in step with the order of the values — which SQLite's
|
|
// positional `?` did for free. Handing out the placeholder and storing the value
|
|
// in one call is what keeps them in step: a filter can be added, removed or
|
|
// reordered without renumbering anything by hand.
|
|
type sqlArgs struct{ vals []any }
|
|
|
|
// add stores v and returns the placeholder that refers to it.
|
|
func (a *sqlArgs) add(v any) string {
|
|
a.vals = append(a.vals, v)
|
|
return "$" + strconv.Itoa(len(a.vals))
|
|
}
|
|
|
|
// addList stores every value and returns their placeholders as "$1, $2, …",
|
|
// ready to drop into an IN (…) clause. Returns an empty string for no values,
|
|
// which no caller should reach: `IN ()` is a syntax error in Postgres as it was
|
|
// in SQLite, so callers check for an empty set before building the query.
|
|
func (a *sqlArgs) addList(vs []any) string {
|
|
parts := make([]string, len(vs))
|
|
for i, v := range vs {
|
|
parts[i] = a.add(v)
|
|
}
|
|
return strings.Join(parts, ", ")
|
|
}
|
|
|
|
// all returns the accumulated values, to be passed straight to Query or Exec.
|
|
func (a *sqlArgs) all() []any { return a.vals }
|
|
|
|
// nowEpoch is the SQL expression for "now, as unix seconds", matching how every
|
|
// timestamp in this schema is stored. SQLite spelled it unixepoch().
|
|
//
|
|
// FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to
|
|
// bigint rounds half up, so a row written at .6 of a second would claim a
|
|
// timestamp one second in the future — off by one against the time.Now().Unix()
|
|
// the Go side stamps, which is what the expiry tests measure.
|
|
const nowEpoch = "FLOOR(EXTRACT(EPOCH FROM now()))::bigint"
|
|
|
|
// isUniqueViolation reports whether err is a broken unique constraint, which
|
|
// callers turn into 409 Conflict rather than 500.
|
|
//
|
|
// Postgres reports it as SQLSTATE 23505 on a typed error; the SQLite driver this
|
|
// replaced only put "UNIQUE constraint failed" in the message, which is why the
|
|
// check used to be a substring match. Matching the code means a renamed
|
|
// constraint or a translated message cannot quietly turn a conflict back into a
|
|
// 500.
|
|
func isUniqueViolation(err error) bool {
|
|
var pgErr *pgconn.PgError
|
|
return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation
|
|
}
|
|
|
|
func respond(w http.ResponseWriter, status int, v any) {
|
|
w.Header().Set("Content-Type", "application/json")
|
|
w.WriteHeader(status)
|
|
json.NewEncoder(w).Encode(v)
|
|
}
|
|
|
|
func decodeJSON(r *http.Request, v any) error {
|
|
defer r.Body.Close()
|
|
return json.NewDecoder(r.Body).Decode(v)
|
|
}
|
|
|
|
func errResp(msg string) map[string]string {
|
|
return map[string]string{"error": msg}
|
|
}
|
|
|
|
// withAdvisoryLock runs fn only if it can take the named Postgres advisory lock on a
|
|
// dedicated connection, and skips fn otherwise. This is what keeps the archiver and
|
|
// notifier safe to run on more than one replica: whichever instance's tick gets there
|
|
// first does the work; the rest see the lock held and simply wait for their next tick
|
|
// instead of running the same pass concurrently.
|
|
//
|
|
// pg_try_advisory_lock is session-scoped, so taking and releasing it must happen on the
|
|
// same connection, reserved via db.Conn rather than borrowed from the pool's shared
|
|
// connections fn itself may use — and released (unlocked, then closed) before returning,
|
|
// since a session lock otherwise outlives this call and leaks onto whatever reuses the
|
|
// pooled connection next.
|
|
func withAdvisoryLock(ctx context.Context, db *sql.DB, key int64, name string, fn func()) {
|
|
conn, err := db.Conn(ctx)
|
|
if err != nil {
|
|
log.Printf("%s: advisory lock: acquire connection: %v", name, err)
|
|
return
|
|
}
|
|
defer conn.Close()
|
|
|
|
var locked bool
|
|
if err := conn.QueryRowContext(ctx, "SELECT pg_try_advisory_lock($1)", key).Scan(&locked); err != nil {
|
|
log.Printf("%s: advisory lock: %v", name, err)
|
|
return
|
|
}
|
|
if !locked {
|
|
return // another replica is already running this pass
|
|
}
|
|
defer func() {
|
|
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", key); err != nil {
|
|
log.Printf("%s: advisory unlock: %v", name, err)
|
|
}
|
|
}()
|
|
|
|
fn()
|
|
}
|