9029d48584
- TERDUT_OPERATOR_KEY creates or re-keys the instance-scoped service account
"terdut-operator" at every start, so terdut-operator needs no bootstrap
handshake. An instance-scoped account now acts as owner of every team's
configuration, but is not a member of any team.
- POST /api/teams takes an external_id (instance service accounts only) and
is idempotent on it, so automation finds its own team again after a crash
instead of adopting by display name. GET /api/teams?name= is removed.
- Integration and dead man's switch names are unique per team (409). The
escalation PUT accepts usernames and resolves them itself.
- The 18 migrations are squashed into 001_schema.sql, with no Default team.
TERDUT_DEADMAN_* and the env seeding of switches are removed: teams carry
their own. Existing development databases must be recreated.
Security and robustness:
- GET /api/users no longer returns other people's email or ntfy topic to
non-admins.
- The access log records the route pattern, so integration keys and ack
tokens in the path are not written to the log. Server errors are logged.
- Rate limits take the client address TERDUT_TRUSTED_PROXIES hops from the
right of X-Forwarded-For instead of trusting the first, forgeable entry.
- /api/bootstrap runs in a transaction under an advisory lock, so two
concurrent calls cannot both create an administrator.
- API key last_used_at is written at most every five minutes.
Cleanup: remove GET /api/incidents/{id}/alerts, unused exports, SQLite
remnants in comments and config.
Claude-Session: https://claude.ai/code/session_016mBLURvJoMuUEr9cB2RpUN
106 lines
3.5 KiB
Go
106 lines
3.5 KiB
Go
package api
|
|
|
|
import (
|
|
"context"
|
|
"crypto/sha256"
|
|
"database/sql"
|
|
"encoding/hex"
|
|
"log"
|
|
"net/http"
|
|
"time"
|
|
|
|
"github.com/go-chi/chi/v5"
|
|
)
|
|
|
|
// issueAckToken mints the secret behind one notification's Acknowledge button
|
|
// and returns the raw value to embed in its URL. Only the hash is stored, the
|
|
// same way api_keys works.
|
|
//
|
|
// A fresh token per delivery rather than one per incident: the raw value only
|
|
// exists for as long as it takes to build the message, so there is nothing to
|
|
// look up and reuse later, and a reminder that supersedes an earlier page
|
|
// carries its own credential.
|
|
func issueAckToken(ctx context.Context, q querier, incidentID, userID int64) (string, error) {
|
|
raw, hash, err := randomToken()
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
now := time.Now()
|
|
if _, err := q.ExecContext(ctx, `
|
|
INSERT INTO incident_ack_tokens (token_hash, incident_id, user_id, created_at, expires_at)
|
|
VALUES ($1, $2, $3, $4, $5)`,
|
|
hash, incidentID, userID, now.Unix(), now.Add(ackTokenTTL).Unix()); err != nil {
|
|
return "", err
|
|
}
|
|
return raw, nil
|
|
}
|
|
|
|
// handleNotifyAck acknowledges an incident from the Acknowledge button in a
|
|
// push notification.
|
|
//
|
|
// It is deliberately outside AuthMiddleware: the caller is a phone acting on a
|
|
// notification, not a client holding an API key. What stands in for the key is
|
|
// the token in the path — 256 bits of entropy, valid for one incident, one
|
|
// action, and one day. It must stay publicly reachable for the button to work
|
|
// when the responder is off the cluster network.
|
|
func handleNotifyAck(db *sql.DB) http.HandlerFunc {
|
|
return func(w http.ResponseWriter, r *http.Request) {
|
|
h := sha256.Sum256([]byte(chi.URLParam(r, "token")))
|
|
hash := hex.EncodeToString(h[:])
|
|
|
|
var incidentID, userID int64
|
|
err := db.QueryRowContext(r.Context(), `
|
|
SELECT incident_id, user_id FROM incident_ack_tokens
|
|
WHERE token_hash = $1 AND expires_at > $2`,
|
|
hash, time.Now().Unix()).Scan(&incidentID, &userID)
|
|
if err != nil {
|
|
// Unknown and expired get the same answer, so the endpoint cannot be
|
|
// used to probe which tokens once existed.
|
|
respond(w, http.StatusNotFound, errResp("invalid or expired token"))
|
|
return
|
|
}
|
|
|
|
acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID)
|
|
if err != nil {
|
|
serverError(w, r, err)
|
|
return
|
|
}
|
|
if !acked {
|
|
// Either the incident closed between the page and the tap, or it was
|
|
// already acknowledged (e.g. from the web UI, or an earlier tap of
|
|
// the same button) — either way nothing the responder did wrong, so
|
|
// report the actual state rather than assuming "resolved", and let
|
|
// ntfy show a success toast rather than a failure.
|
|
inc, err := fetchIncident(r.Context(), db, incidentID)
|
|
if err != nil {
|
|
serverError(w, r, err)
|
|
return
|
|
}
|
|
respond(w, http.StatusOK, map[string]any{
|
|
"incident_id": incidentID,
|
|
"status": inc.Status,
|
|
})
|
|
return
|
|
}
|
|
respond(w, http.StatusOK, map[string]any{
|
|
"incident_id": incidentID,
|
|
"status": "acknowledged",
|
|
})
|
|
}
|
|
}
|
|
|
|
// purgeAckTokens drops tokens whose notifications are long past. Nothing else
|
|
// deletes them: incidents are archived rather than removed, so the cascade never
|
|
// fires in practice.
|
|
func purgeAckTokens(ctx context.Context, db *sql.DB) {
|
|
res, err := db.ExecContext(ctx,
|
|
"DELETE FROM incident_ack_tokens WHERE expires_at < $1", time.Now().Unix())
|
|
if err != nil {
|
|
log.Printf("sweeper: purge ack tokens: %v", err)
|
|
return
|
|
}
|
|
if n, _ := res.RowsAffected(); n > 0 {
|
|
log.Printf("sweeper: purged %d expired ack token(s)", n)
|
|
}
|
|
}
|