Files
terdut-server/internal/api/notify_ack.go
T
Niklas Ye 9029d48584
CI / chart (pull_request) Successful in 2s
CI / security (pull_request) Failing after 19s
CI / test (pull_request) Successful in 5m34s
Let an operator authenticate with a seeded key, and reset the schema
- TERDUT_OPERATOR_KEY creates or re-keys the instance-scoped service account
  "terdut-operator" at every start, so terdut-operator needs no bootstrap
  handshake. An instance-scoped account now acts as owner of every team's
  configuration, but is not a member of any team.
- POST /api/teams takes an external_id (instance service accounts only) and
  is idempotent on it, so automation finds its own team again after a crash
  instead of adopting by display name. GET /api/teams?name= is removed.
- Integration and dead man's switch names are unique per team (409). The
  escalation PUT accepts usernames and resolves them itself.
- The 18 migrations are squashed into 001_schema.sql, with no Default team.
  TERDUT_DEADMAN_* and the env seeding of switches are removed: teams carry
  their own. Existing development databases must be recreated.

Security and robustness:
- GET /api/users no longer returns other people's email or ntfy topic to
  non-admins.
- The access log records the route pattern, so integration keys and ack
  tokens in the path are not written to the log. Server errors are logged.
- Rate limits take the client address TERDUT_TRUSTED_PROXIES hops from the
  right of X-Forwarded-For instead of trusting the first, forgeable entry.
- /api/bootstrap runs in a transaction under an advisory lock, so two
  concurrent calls cannot both create an administrator.
- API key last_used_at is written at most every five minutes.

Cleanup: remove GET /api/incidents/{id}/alerts, unused exports, SQLite
remnants in comments and config.

Claude-Session: https://claude.ai/code/session_016mBLURvJoMuUEr9cB2RpUN
2026-10-09 14:56:13 +02:00

106 lines
3.5 KiB
Go

package api
import (
"context"
"crypto/sha256"
"database/sql"
"encoding/hex"
"log"
"net/http"
"time"
"github.com/go-chi/chi/v5"
)
// issueAckToken mints the secret behind one notification's Acknowledge button
// and returns the raw value to embed in its URL. Only the hash is stored, the
// same way api_keys works.
//
// A fresh token per delivery rather than one per incident: the raw value only
// exists for as long as it takes to build the message, so there is nothing to
// look up and reuse later, and a reminder that supersedes an earlier page
// carries its own credential.
func issueAckToken(ctx context.Context, q querier, incidentID, userID int64) (string, error) {
raw, hash, err := randomToken()
if err != nil {
return "", err
}
now := time.Now()
if _, err := q.ExecContext(ctx, `
INSERT INTO incident_ack_tokens (token_hash, incident_id, user_id, created_at, expires_at)
VALUES ($1, $2, $3, $4, $5)`,
hash, incidentID, userID, now.Unix(), now.Add(ackTokenTTL).Unix()); err != nil {
return "", err
}
return raw, nil
}
// handleNotifyAck acknowledges an incident from the Acknowledge button in a
// push notification.
//
// It is deliberately outside AuthMiddleware: the caller is a phone acting on a
// notification, not a client holding an API key. What stands in for the key is
// the token in the path — 256 bits of entropy, valid for one incident, one
// action, and one day. It must stay publicly reachable for the button to work
// when the responder is off the cluster network.
func handleNotifyAck(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
h := sha256.Sum256([]byte(chi.URLParam(r, "token")))
hash := hex.EncodeToString(h[:])
var incidentID, userID int64
err := db.QueryRowContext(r.Context(), `
SELECT incident_id, user_id FROM incident_ack_tokens
WHERE token_hash = $1 AND expires_at > $2`,
hash, time.Now().Unix()).Scan(&incidentID, &userID)
if err != nil {
// Unknown and expired get the same answer, so the endpoint cannot be
// used to probe which tokens once existed.
respond(w, http.StatusNotFound, errResp("invalid or expired token"))
return
}
acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID)
if err != nil {
serverError(w, r, err)
return
}
if !acked {
// Either the incident closed between the page and the tap, or it was
// already acknowledged (e.g. from the web UI, or an earlier tap of
// the same button) — either way nothing the responder did wrong, so
// report the actual state rather than assuming "resolved", and let
// ntfy show a success toast rather than a failure.
inc, err := fetchIncident(r.Context(), db, incidentID)
if err != nil {
serverError(w, r, err)
return
}
respond(w, http.StatusOK, map[string]any{
"incident_id": incidentID,
"status": inc.Status,
})
return
}
respond(w, http.StatusOK, map[string]any{
"incident_id": incidentID,
"status": "acknowledged",
})
}
}
// purgeAckTokens drops tokens whose notifications are long past. Nothing else
// deletes them: incidents are archived rather than removed, so the cascade never
// fires in practice.
func purgeAckTokens(ctx context.Context, db *sql.DB) {
res, err := db.ExecContext(ctx,
"DELETE FROM incident_ack_tokens WHERE expires_at < $1", time.Now().Unix())
if err != nil {
log.Printf("sweeper: purge ack tokens: %v", err)
return
}
if n, _ := res.RowsAffected(); n > 0 {
log.Printf("sweeper: purged %d expired ack token(s)", n)
}
}