package api_test import ( "bytes" "encoding/json" "fmt" "net/http" "sync" "testing" ) // TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident reproduces two // replicas racing the very first webhook delivery for a brand-new group_key: // both see no open incident yet (incidentForGroup's own SELECT finds // nothing) and race openIncident's INSERT. // // The DB's own unique index already guarantees at most one incident either // way, with or without this fix — so "exactly one incident" alone cannot // tell a fixed run from a broken one. What ON CONFLICT handling actually // changes is what happens to the *loser*: before it, the loser's INSERT hit // incidents_open_group_key_idx's unique violation, which — since // upsertAlerts ran earlier in that same transaction — rolled back its whole // payload, alert insert included. ingest's error is only logged and // receiveWebhook answers 200 regardless, so nothing ever retried it: the // loser's alert silently never existed. That is the regression signal this // test checks — every caller's fingerprint must show up in /api/alerts, not // just the winner's. func TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident(t *testing.T) { s := newTS(t) const callers = 8 const groupKey = "race-group" // Every caller needs its own fingerprint. A shared one would serialize all // of them at upsertAlerts' own ON CONFLICT (team_id, fingerprint) row lock, // long before any of them reached incidentForGroup — which would hide the // very race this test exists to force. bodies := make([][]byte, callers) for i := range callers { payload := map[string]any{ "version": "4", "status": "firing", "groupKey": groupKey, "groupLabels": map[string]string{"alertname": "RaceAlert"}, "alerts": []map[string]any{amAlert(fmt.Sprintf("fp-race-%d", i), "RaceAlert", "firing", "2026-05-20T10:00:00Z", "0001-01-01T00:00:00Z", nil)}, } bodies[i], _ = json.Marshal(payload) } // A start line, so every request is fired as close to simultaneously as // goroutine scheduling allows, rather than trickling out one dial at a // time — the race window is the gap between incidentForGroup's SELECT and // openIncident's INSERT, which a staggered start could easily miss. var ready sync.WaitGroup start := make(chan struct{}) statuses := make([]int, callers) var wg sync.WaitGroup for i := range callers { ready.Add(1) wg.Add(1) go func(i int) { defer wg.Done() ready.Done() <-start resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager", "application/json", bytes.NewReader(bodies[i])) if err != nil { t.Errorf("post webhook #%d: %v", i, err) return } defer resp.Body.Close() statuses[i] = resp.StatusCode }(i) } ready.Wait() close(start) wg.Wait() for i, code := range statuses { if code != http.StatusOK { t.Errorf("webhook #%d returned %d, want 200", i, code) } } var matched []any for _, inc := range listIncidents(t, s, "") { if inc["group_key"] == groupKey { matched = append(matched, inc["id"]) } } if len(matched) != 1 { t.Fatalf("expected exactly 1 incident for group_key %q after %d concurrent deliveries, got %d: %v", groupKey, callers, len(matched), matched) } var alerts []map[string]any decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts) seen := map[string]bool{} for _, a := range alerts { if fp, ok := a["fingerprint"].(string); ok { seen[fp] = true } } for i := range callers { fp := fmt.Sprintf("fp-race-%d", i) if !seen[fp] { t.Errorf("alert %q is missing: its delivery's whole payload was silently rolled back "+ "when it lost the race for the incident", fp) } } }