This commit is contained in:
binwiederhier
2026-08-02 23:54:35 +02:00
parent 5a6c4277ad
commit ccfbc2309d
14 changed files with 459 additions and 221 deletions
+17 -13
View File
@@ -370,11 +370,14 @@ func New(conf *Config) (*Server, error) {
// be reached from the outside even before any firewalling.
func (s *Server) clusterHandler() http.Handler {
mux := http.NewServeMux()
// TODO(T8): reflect s.cluster.Healthy() here (and in handleHealth) instead of a static
// response, so health checks can pull a node that lost its registry heartbeat
mux.HandleFunc(apiHealthPath, func(w http.ResponseWriter, _ *http.Request) {
w.Header().Set("Content-Type", "application/json")
io.WriteString(w, `{"healthy":true}`+"\n")
if s.cluster.Healthy() {
io.WriteString(w, `{"healthy":true}`+"\n")
} else {
w.WriteHeader(http.StatusServiceUnavailable)
io.WriteString(w, `{"healthy":false}`+"\n")
}
})
mux.Handle("/", s.cluster)
return mux
@@ -396,15 +399,15 @@ func (s *Server) liveTopics() []string {
}
// topicAnnouncer returns the first-subscriber hook for a topic: it tells peer nodes right away
// that this node now wants messages for it (see Cluster.AnnounceTopics).
// that this node now wants messages for it (see Cluster.BroadcastState).
func (s *Server) topicAnnouncer(id string) func() {
return func() {
s.cluster.AnnounceTopics([]string{id})
s.cluster.BroadcastState(&cluster.State{AddedTopics: []string{id}})
}
}
// deliverFromBus delivers a message received from a peer node (via the cluster) to this
// node's local subscribers. It is the receive-side counterpart to Cluster.Relay: local
// node's local subscribers. It is the receive-side counterpart to Cluster.ForwardMessage: local
// delivery and all global side effects (Firebase, email, web push, upstream) already ran on the
// origin node, so this only publishes to the local topic, and never re-relays.
func (s *Server) deliverFromBus(m *model.Message) {
@@ -829,12 +832,13 @@ func (s *Server) handleTopicAuth(w http.ResponseWriter, _ *http.Request, _ *visi
}
func (s *Server) handleHealth(w http.ResponseWriter, _ *http.Request, _ *visitor) error {
// TODO(T8): in cluster mode, reflect s.cluster.Healthy() so DNS/LB health checks stop
// routing NEW clients to a node whose peers no longer relay to it (see Cluster interface)
response := &apiHealthResponse{
Healthy: true,
// Unhealthy = the registry heartbeat went stale and peers stopped forwarding to this
// node; 503 lets status-code LB checks pull it (checkers must fail open, see cluster.Cluster)
healthy := s.cluster.Healthy()
if !healthy {
w.WriteHeader(http.StatusServiceUnavailable)
}
return s.writeJSON(w, response)
return s.writeJSON(w, &apiHealthResponse{Healthy: healthy})
}
// handleMetrics returns Prometheus metrics. This endpoint is only called if enable-metrics is set,
@@ -965,8 +969,8 @@ func (s *Server) dispatch(v *visitor, t *topic, m *model.Message, opts dispatchO
return err
}
}
// Relay to peer cluster nodes, whose subscribers do not show up in this node's topics map
if err := s.cluster.Relay(m); err != nil {
// ForwardMessage to peer cluster nodes, whose subscribers do not show up in this node's topics map
if err := s.cluster.ForwardMessage(m); err != nil {
logvm(v, m).Err(err).Warn("Cluster: unable to relay message to peer nodes")
}
// Fire the requested side-effect targets
+44 -9
View File
@@ -12,6 +12,7 @@ import (
"time"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/cluster"
dbtest "heckel.io/ntfy/v2/db/test"
"heckel.io/ntfy/v2/model"
"heckel.io/ntfy/v2/user"
@@ -20,13 +21,14 @@ import (
// fakeCluster records relayed messages and topic announcements so tests can assert that every
// publish path passes through the cluster exactly once, and that subscription hooks fire.
type fakeCluster struct {
mu sync.Mutex
messages []*model.Message
announced []string
notLeader bool
mu sync.Mutex
messages []*model.Message
announced []string
notLeader bool
notHealthy bool
}
func (b *fakeCluster) Relay(m *model.Message) error {
func (b *fakeCluster) ForwardMessage(m *model.Message) error {
b.mu.Lock()
defer b.mu.Unlock()
b.messages = append(b.messages, m)
@@ -35,10 +37,22 @@ func (b *fakeCluster) Relay(m *model.Message) error {
func (b *fakeCluster) ServeHTTP(_ http.ResponseWriter, _ *http.Request) {}
func (b *fakeCluster) AnnounceTopics(topics []string) {
func (b *fakeCluster) BroadcastState(state *cluster.State) {
b.mu.Lock()
defer b.mu.Unlock()
b.announced = append(b.announced, topics...)
b.announced = append(b.announced, state.AddedTopics...)
}
func (b *fakeCluster) Healthy() bool {
b.mu.Lock()
defer b.mu.Unlock()
return !b.notHealthy
}
func (b *fakeCluster) setHealthy(healthy bool) {
b.mu.Lock()
defer b.mu.Unlock()
b.notHealthy = !healthy
}
func (b *fakeCluster) IsLeader() bool {
@@ -67,7 +81,7 @@ func (b *fakeCluster) Announced() []string {
return append([]string{}, b.announced...)
}
func TestServer_Cluster_PublishRelaysOnce(t *testing.T) {
func TestServer_Cluster_PublishForwardsOnce(t *testing.T) {
s := newTestServer(t, newTestConfig(t, ""))
b := &fakeCluster{}
s.cluster = b
@@ -79,7 +93,7 @@ func TestServer_Cluster_PublishRelaysOnce(t *testing.T) {
require.Equal(t, "hi there", messages[0].Message)
}
func TestServer_Cluster_SyncEventRelays(t *testing.T) {
func TestServer_Cluster_SyncEventForwards(t *testing.T) {
// Account sync events are delivered via the user's st_... sync topic; without relaying
// them, cross-device account sync silently breaks when a user's devices land on different
// cluster nodes.
@@ -316,3 +330,24 @@ func TestServer_Cluster_FirebaseKeepaliverOnlyOnLeader(t *testing.T) {
cl.setLeader(true)
waitFor(t, func() bool { return len(sender.Messages()) > 0 })
}
func TestServer_Cluster_HealthReflectsCluster(t *testing.T) {
// A node whose registry heartbeat went stale no longer receives forwarded messages, so
// health checks must pull it from rotation (the fail-open policy lives in the checker)
s := newTestServer(t, newTestConfig(t, ""))
cl := &fakeCluster{}
s.cluster = cl
rr := request(t, s, "GET", "/v1/health", "", nil)
require.Equal(t, 200, rr.Code)
require.Contains(t, rr.Body.String(), `"healthy":true`)
cl.setHealthy(false)
rr = request(t, s, "GET", "/v1/health", "", nil)
require.Equal(t, 503, rr.Code)
require.Contains(t, rr.Body.String(), `"healthy":false`)
// The cluster listener's health endpoint reflects the same state
rr2 := httptest.NewRecorder()
req, err := http.NewRequest("GET", "/v1/health", nil)
require.Nil(t, err)
s.clusterHandler().ServeHTTP(rr2, req)
require.Equal(t, 503, rr2.Code)
}