Compare commits

..
Author SHA1 Message Date
binwiederhier 81a589483f Merge branch 'main' into cluster2 2026-08-04 09:08:46 +02:00
binwiederhier ccfbc2309d Refactor 2026-08-02 23:54:35 +02:00
binwiederhier 5a6c4277ad WIP: Clsuter support (cross node delivery, leader election) 2026-08-01 16:10:31 +02:00
54 changed files with 4144 additions and 1031 deletions
+1 -1
View File
@@ -1 +1 @@
1.27.0
1.26.5
+113
View File
@@ -0,0 +1,113 @@
// Package cluster implements cross-node message delivery for a multi-node ntfy cluster. Nodes
// register themselves in a PostgreSQL node registry (control plane) and fan published messages
// out to each other directly over HTTP (data plane); PostgreSQL is never on the message path.
// The single-node default is the nop cluster, which does nothing.
package cluster
import (
"errors"
"net/http"
"time"
"heckel.io/ntfy/v2/db"
"heckel.io/ntfy/v2/model"
)
// The internal peer API: every kind of node-to-node communication is a path under
// /v1/internal/, served only on the dedicated cluster listener. Future concerns (rate limit
// counters, stats) become new paths or new sections of the state envelope.
const (
// MessagePath receives batches of published messages (NDJSON, one apiMessage per line).
MessagePath = "/v1/internal/message"
// StatePath receives peer state (JSON apiState): full subscription snapshots and
// incremental updates.
StatePath = "/v1/internal/state"
)
// NodeID identifies a cluster node; it keys the registry, the per-peer queues, and the peer
// state table.
//
// Naming convention: a "node" is any cluster member in the absolute sense (identity, registry,
// config); a "peer" is another node as seen from this one (Peers, peerQueue, peerState). A peer
// IS a node, which is why peer values carry a NodeID.
type NodeID string
const (
// secretHeader carries the shared secret authenticating node-to-node fan-out requests.
secretHeader = "X-Cluster-Secret"
// originHeader carries the sending node's ID on fan-out requests, so a node can skip
// requests that carry its own broadcasts (loop prevention).
originHeader = "X-Cluster-Origin"
)
// Content types of the peer API: message bodies are NDJSON (one JSON message per line, matching
// the framing of ntfy's own /topic/json subscribe stream), state bodies are plain JSON. Future
// node-to-node request types get their own paths on the cluster listener; an old node answering
// 404 on an unknown path keeps mixed-version clusters working during rolling deploys.
const (
contentTypeNDJSON = "application/x-ndjson"
contentTypeJSON = "application/json"
)
const (
defaultHeartbeatInterval = 3 * time.Second // How often a node refreshes its registry heartbeat
defaultNodeTTL = 30 * time.Second // A node counts as live if its heartbeat is newer than this; generous to avoid false-dead flapping (see plans)
defaultStateInterval = 15 * time.Second // How often the full subscription state is pushed to peers
// DefaultBatchLinger is how long a fan-out message may wait in a peer's queue for more
// messages to arrive, so they are delivered as one batch. It trades up to this much
// cross-node latency for a bounded request rate per peer.
DefaultBatchLinger = 500 * time.Millisecond
)
// Cluster fans published messages out to peer cluster nodes and receives their fan-out requests.
// Local delivery to a node's own subscribers still happens inline in the server; the cluster
// only covers the cross-node hop.
type Cluster interface {
http.Handler
// ForwardMessage sends a locally published message on to the peer nodes that may have subscribers
// for its topic (all of them, when subscription knowledge is missing or stale). It is
// fire-and-forget and must not block the caller's request path.
ForwardMessage(m *model.Message) error
// BroadcastState pushes a subscription-state delta to ALL peers (unlike ForwardMessage,
// which routes), closing the routing-knowledge window to ~one round trip. Nop single-node.
BroadcastState(state *State)
// IsLeader reports whether this node holds the cluster leader lock. Singleton background
// jobs (e.g. the Firebase keepaliver) are gated on the leader.
IsLeader() bool
// Healthy reports whether this node is fit to serve: its registry heartbeat is fresh
// enough (within NodeTTL) that peers still forward messages to it. Health checkers must
// fail open (never pull ALL nodes): during a full database outage every node reports
// unhealthy while the mesh keeps delivering on stale peer caches.
Healthy() bool
// Close stops the cluster and releases its resources.
Close() error
}
// New creates the cluster for the given config: the nop cluster when clustering is disabled (the
// single-node default), or the peer-mesh cluster otherwise.
func New(conf *Config, pool *db.DB, deliver DeliverFunc, topics TopicsFunc) (Cluster, error) {
if !conf.Enabled {
return &nopCluster{}, nil
}
if pool == nil {
return nil, errors.New("cluster mode requires a PostgreSQL database (set database-url)")
}
if conf.AdvertiseURL == "" {
return nil, errors.New("cluster mode requires an advertise URL (set cluster-advertise-url)")
}
if conf.NodeID == "" {
return nil, errors.New("cluster mode requires a stable node ID (set cluster-node-id)")
}
if conf.HeartbeatInterval == 0 {
conf.HeartbeatInterval = defaultHeartbeatInterval
}
if conf.NodeTTL == 0 {
conf.NodeTTL = defaultNodeTTL
}
if conf.StateInterval == 0 {
conf.StateInterval = defaultStateInterval
}
return newMeshCluster(conf, pool, deliver, topics)
}
+133
View File
@@ -0,0 +1,133 @@
package cluster
import (
"fmt"
"io"
"net/http"
"net/http/httptest"
"os"
"sync"
"testing"
"time"
"github.com/stretchr/testify/require"
dbtest "heckel.io/ntfy/v2/db/test"
"heckel.io/ntfy/v2/model"
)
// TestMesh_Soak floods the mesh with concurrent publishers and asserts exact delivery: every
// message reaches the peer exactly once, nothing is dropped, and batching keeps the request
// count far below the message count. Skipped unless NTFY_TEST_SOAK is set (it takes a few
// seconds and is meant for pre-deploy verification, not the regular suite).
func TestMesh_Soak(t *testing.T) {
if os.Getenv("NTFY_TEST_SOAK") == "" {
t.Skip("NTFY_TEST_SOAK not set")
}
// ~1000 msg/s aggregate (10x the ntfy.sh peak of ~88 msg/s): each publisher paces itself to
// 100 msg/s. Unthrottled publishing intentionally overruns the bounded per-peer queue (load
// shedding by design), so a zero-drop assertion only holds below the drain ceiling.
const (
publishers = 10
messagesPerPublisher = 300
publishInterval = 10 * time.Millisecond
total = publishers * messagesPerPublisher
)
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var mu sync.Mutex
received := make(map[string]int, total) // message body -> count, to catch duplicates
requests := 0
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(r.Body)
require.Nil(t, err)
messages, err := unmarshalMessageBody(body, 1<<20)
require.Nil(t, err)
mu.Lock()
requests++
for _, m := range messages {
received[m.Message]++
}
mu.Unlock()
w.WriteHeader(http.StatusOK)
}))
defer srv.Close()
conf := newTestMeshConfig("node-a", "http://127.0.0.1:1")
conf.BatchLinger = 50 * time.Millisecond
conf.NodeTTL = time.Minute // The fake peer never heartbeats; liveness is not under test here
registerFakePeer(t, pool, "node-peer", srv.URL)
mesh, err := newMeshCluster(conf, pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
start := time.Now()
var wg sync.WaitGroup
for p := 0; p < publishers; p++ {
wg.Add(1)
go func(p int) {
defer wg.Done()
ticker := time.NewTicker(publishInterval)
defer ticker.Stop()
for i := 0; i < messagesPerPublisher; i++ {
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("mytopic", fmt.Sprintf("p%d-m%d", p, i))))
<-ticker.C
}
}(p)
}
wg.Wait()
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return len(received) == total
})
elapsed := time.Since(start)
mu.Lock()
defer mu.Unlock()
for body, count := range received {
require.Equalf(t, 1, count, "message %s delivered %d times", body, count)
}
require.Less(t, requests, total/10, "expected strong batching under load")
t.Logf("soak: %d messages, %d requests (%.1f msgs/request), %.0f msgs/s",
total, requests, float64(total)/float64(requests), float64(total)/elapsed.Seconds())
}
// BenchmarkForwardMessage measures the publish-path cost of ForwardMessage: marshal + peer lookup (cached)
// + enqueue. The peer never drains, so enqueued fragments are dropped once the queue fills;
// the benchmark measures the hot path, not HTTP delivery.
func BenchmarkForwardMessage(b *testing.B) {
if os.Getenv("NTFY_TEST_DATABASE_URL") == "" {
b.Skip("NTFY_TEST_DATABASE_URL not set")
}
schemaDSN := dbtest.CreateTestPostgresSchema(b)
pool := openTestPool(b, schemaDSN)
conf := newTestMeshConfig("node-a", "http://127.0.0.1:1")
conf.BatchLinger = time.Minute // Never flush; we measure enqueue only
mesh, err := newMeshCluster(conf, pool, nil, nil)
require.Nil(b, err)
defer mesh.Close()
registerFakePeer(b, pool, "node-peer", "http://127.0.0.1:1")
m := model.NewDefaultMessage("mytopic", "benchmark message body of typical size for a push")
b.ResetTimer()
for i := 0; i < b.N; i++ {
if err := mesh.ForwardMessage(m); err != nil {
b.Fatal(err)
}
}
}
// BenchmarkDecodeFanout measures the receive-path cost of decoding a 100-message NDJSON body.
func BenchmarkDecodeFanout(b *testing.B) {
frags := make([][]byte, 100)
for i := range frags {
frag, err := marshalMessage(model.NewDefaultMessage("mytopic", fmt.Sprintf("benchmark message %d", i)))
require.Nil(b, err)
frags[i] = frag
}
body := assembleMessageBody(frags)
b.SetBytes(int64(len(body)))
b.ResetTimer()
for i := 0; i < b.N; i++ {
messages, err := unmarshalMessageBody(body, 1<<20)
if err != nil || len(messages) != 100 {
b.Fatal("decode failed")
}
}
}
+498
View File
@@ -0,0 +1,498 @@
package cluster
import (
"bytes"
"context"
"crypto/subtle"
"encoding/json"
"io"
"net/http"
"sync"
"time"
"heckel.io/ntfy/v2/cluster/registry"
"heckel.io/ntfy/v2/db"
"heckel.io/ntfy/v2/db/pg"
"heckel.io/ntfy/v2/log"
"heckel.io/ntfy/v2/metrics"
"heckel.io/ntfy/v2/model"
"heckel.io/ntfy/v2/util"
)
const (
meshHTTPTimeout = 5 * time.Second
peerQueueSize = 1024 // Bounded per-peer fan-out queue (drop on overflow)
batchMaxMessages = 100 // Flush a batch early when it reaches this many messages
batchMaxBytes = 256 * 1024 // Flush a batch early when it reaches this size
stateMaxBytes = 4 * 1024 * 1024 // Upper bound for inbound state bodies (filter over ~1M topics)
stateFilterFPRate = 0.01 // Bloom false-positive rate; a false positive is one wasted send
tag = "cluster"
)
// meshCluster fans messages out directly to peer nodes over HTTP (the data plane), using
// PostgreSQL only as a control plane: the node_registry table for membership/discovery, and a
// Postgres advisory lock for singleton-job leader election. Fan-out never touches the database on
// the message path (only the cached peer list does). See plans/260715-scale-out-mesh.md.
//
// Each peer has its own bounded send queue and delivery worker, so a slow or wedged peer only
// backs up (and eventually drops) its own queue and never delays delivery to healthy peers.
type meshCluster struct {
conf *Config
deliver DeliverFunc
topics TopicsFunc
registry *registry.Registry
leader *pg.Leader
httpClient *http.Client
mux *http.ServeMux // The internal peer API; Cluster is an http.Handler
queues map[NodeID]*peerQueue // per-peer send queues; reconciled against the registry
closed bool // Guards against ForwardMessage spawning new workers after Close
states map[NodeID]*peerState // what each peer last told us (subscription knowledge)
lastStatePush time.Time // Only touched by the heartbeat goroutine
knownPeers map[NodeID]string // Peers seen in the last reconcile, for join/leave logging
lastRegistered time.Time // Last successful registry heartbeat, for Healthy
ctx context.Context
cancel context.CancelFunc
wg sync.WaitGroup
mu sync.Mutex // Protects queues, closed, knownPeers and lastRegistered
statesMu sync.Mutex // Protects states
}
// newMeshCluster creates the mesh cluster: it sets up the registry schema, registers this node
// (synchronously, so it is discoverable before New returns), and starts the heartbeat loop.
// Peer delivery workers are started lazily as peers appear in the registry.
func newMeshCluster(conf *Config, pool *db.DB, deliver DeliverFunc, topics TopicsFunc) (*meshCluster, error) {
if topics == nil {
topics = func() []string { return nil } // No known topics; peers will broadcast to us
}
reg, err := registry.New(pool, string(conf.NodeID), conf.AdvertiseURL, conf.NodeTTL)
if err != nil {
return nil, err
}
// Register synchronously so the node is discoverable before the constructor returns; the
// heartbeat loop refreshes the registration from here on
if err := reg.Register(); err != nil {
return nil, err
}
ctx, cancel := context.WithCancel(context.Background())
c := &meshCluster{
conf: conf,
deliver: deliver,
topics: topics,
registry: reg,
// Renews its lease on its own fixed cadence; see pg.Leader for the semantics
leader: pg.NewLeader(pool.Primary(), pg.LeaderLockKey, conf.LeaderRenewInterval),
httpClient: &http.Client{Timeout: meshHTTPTimeout},
queues: make(map[NodeID]*peerQueue),
lastRegistered: time.Now(), // The synchronous Register above just succeeded
states: make(map[NodeID]*peerState),
knownPeers: make(map[NodeID]string),
ctx: ctx,
cancel: cancel,
}
c.mux = http.NewServeMux()
c.mux.HandleFunc("POST "+MessagePath, c.authenticated(c.handleMessage))
c.mux.HandleFunc("POST "+StatePath, c.authenticated(c.handleState))
c.wg.Add(1)
go c.heartbeatLoop()
return c, nil
}
// ServeHTTP serves the internal peer API. Auth lives in the authenticated middleware, so every
// endpoint gets the same shared-secret and origin handling.
func (c *meshCluster) ServeHTTP(w http.ResponseWriter, r *http.Request) {
c.mux.ServeHTTP(w, r)
}
// authenticated wraps a peer API handler with the checks every endpoint needs: the shared
// secret (constant-time compare, rejected before any body is read), a present origin, and the
// origin self-skip (a request carrying this node's own traffic is acknowledged but ignored).
func (c *meshCluster) authenticated(h func(origin NodeID, w http.ResponseWriter, r *http.Request)) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
if c.conf.Secret == "" || subtle.ConstantTimeCompare([]byte(r.Header.Get(secretHeader)), []byte(c.conf.Secret)) != 1 {
w.WriteHeader(http.StatusUnauthorized)
return
}
origin := NodeID(r.Header.Get(originHeader))
if origin == "" {
w.WriteHeader(http.StatusBadRequest)
return
}
if origin == c.conf.NodeID {
w.WriteHeader(http.StatusOK) // Our own traffic; nothing to do
return
}
h(origin, w, r)
}
}
// heartbeatLoop runs one heartbeat immediately (the ticker first fires a full interval after
// startup, and a fresh node should be leader-capable and state-visible right away), then one per
// interval until shutdown.
func (c *meshCluster) heartbeatLoop() {
defer c.wg.Done()
ticker := time.NewTicker(c.conf.HeartbeatInterval)
defer ticker.Stop()
if err := c.heartbeat(); err != nil {
log.Tag(tag).Err(err).Warn("Cluster heartbeat failed")
}
for {
select {
case <-c.ctx.Done():
return
case <-ticker.C:
if err := c.heartbeat(); err != nil {
log.Tag(tag).Err(err).Warn("Cluster heartbeat failed")
}
}
}
}
// heartbeat is one control-plane tick: refresh this node's registry row, retry/confirm the
// leader lock, prune long-dead registry rows (as leader), reconcile the per-peer queues, and
// periodically push our subscription state to peers.
//
// A node that cannot even register itself aborts the tick: the remaining database work would
// fail against the same database, and everything downstream degrades safely without it -- ForwardMessage
// serves the stale peer cache on its own, and peers fall back to broadcasting to us once our
// last pushed state expires.
func (c *meshCluster) heartbeat() error {
if err := c.registry.Register(); err != nil {
return err
}
c.mu.Lock()
c.lastRegistered = time.Now()
c.mu.Unlock()
// Effective leadership: pg.Leader's lease semantics guarantee a no-leader gap on
// failover, never two leaders
if c.leader.IsLeader() {
metrics.ClusterLeader.Set(1)
if err := c.registry.Prune(); err != nil {
log.Tag(tag).Err(err).Warn("Failed to prune stale nodes") // Housekeeping only; not fatal for the tick
}
} else {
metrics.ClusterLeader.Set(0)
}
peers, err := c.registry.Peers()
if err != nil {
return err
}
c.reconcilePeers(peers)
if time.Since(c.lastStatePush) >= c.conf.StateInterval {
c.pushState(peers)
c.lastStatePush = time.Now()
}
return nil
}
// reconcilePeers aligns this node's per-peer attachments with the live peer set: it retires the
// queues (and workers) of peers that have left the registry or re-registered under a new
// advertise URL (the retired queue's remainder was headed for a dead address anyway), and prunes
// the stale state of departed peers. New and replacement queues are created lazily by ForwardMessage, not
// here, so a freshly joined peer is reachable immediately.
func (c *meshCluster) reconcilePeers(peers []*registry.Peer) {
metrics.ClusterPeers.Set(float64(len(peers)))
alive := make(map[NodeID]string, len(peers)) // node ID -> advertise URL
for _, p := range peers {
alive[NodeID(p.NodeID)] = p.AdvertiseURL
}
c.mu.Lock()
// Log joins and leaves (as seen through the up-to-NodeTTL-stale registry view)
for nodeID, url := range alive {
if _, ok := c.knownPeers[nodeID]; !ok {
log.Tag(tag).Info("Peer %s (%s) joined the cluster", nodeID, url)
}
}
for nodeID := range c.knownPeers {
if _, ok := alive[nodeID]; !ok {
log.Tag(tag).Info("Peer %s left the cluster", nodeID)
}
}
c.knownPeers = alive
for nodeID, q := range c.queues {
if url, ok := alive[nodeID]; !ok || q.advertiseURL != url {
q.queue.Close() // Flushes the remainder; the worker exits when the queue is drained
delete(c.queues, nodeID)
}
}
c.mu.Unlock()
// Prune the state of departed peers, but only once stale: state is push-driven and can
// arrive before a new peer is visible in the (up to NodeTTL stale) registry view, so fresh
// state must survive even when its peer is not in the live set. Without this, the states of
// long-gone nodes would accumulate forever.
c.statesMu.Lock()
for nodeID, state := range c.states {
if _, ok := alive[nodeID]; !ok && time.Since(state.updatedAt) > 3*c.conf.StateInterval {
delete(c.states, nodeID)
}
}
c.statesMu.Unlock()
}
// queueFor returns the send queue for the given peer, creating it (and its delivery worker) if it
// does not exist yet. The caller must hold c.mu.
func (c *meshCluster) queueFor(p *registry.Peer) *peerQueue {
nodeID := NodeID(p.NodeID)
q, ok := c.queues[nodeID]
if ok {
return q
}
q = &peerQueue{
advertiseURL: p.AdvertiseURL,
queue: util.NewLingerQueue(peerQueueSize, batchMaxMessages, batchMaxBytes,
func(frag []byte) int { return len(frag) }, c.conf.BatchLinger),
}
c.queues[nodeID] = q
c.wg.Add(1)
go c.peerWorker(nodeID, q)
return q
}
// ForwardMessage enqueues the message for delivery to every live peer node that may have subscribers for
// its topic (all of them, absent fresh knowledge). Delivery is fire-and-forget via each peer's
// bounded batching queue; if a peer's queue is full the message is dropped for that peer
// (subscribers reconnect and re-poll history from the database).
func (c *meshCluster) ForwardMessage(msg *model.Message) error {
peers, err := c.registry.Peers()
if err != nil {
return err
}
if len(peers) == 0 {
return nil // Cluster of one; skip the marshal
}
frag, err := marshalMessage(msg)
if err != nil {
return err
}
metrics.ClusterMessagesForwarded.Inc()
c.mu.Lock()
defer c.mu.Unlock()
if c.closed {
return nil // Shutting down; the message is dropped like any other in-flight fan-out
}
for _, p := range peers {
// Route around peers whose fresh state provably excludes this topic; anything less
// certain (no state, stale state) falls back to broadcasting
if !c.mayNeed(NodeID(p.NodeID), msg.Topic) {
metrics.ClusterRouteSkipped.Inc()
if ev := log.Tag(tag); ev.IsTrace() {
ev.Trace("Skipping peer %s for message %s: no subscribers for topic %s", p.NodeID, msg.ID, msg.Topic)
}
continue
}
if !c.queueFor(p).queue.TryEnqueue(frag) {
metrics.ClusterQueueDropped.Inc()
log.Tag(tag).Warn("Fan-out queue for peer %s full, dropping message %s", p.NodeID, msg.ID)
} else if ev := log.Tag(tag); ev.IsTrace() {
ev.Trace("Enqueued message %s (topic %s) for peer %s", msg.ID, msg.Topic, p.NodeID)
}
}
return nil
}
// mayNeed reports whether the peer may have a subscriber for the topic. Conservative by
// construction: it returns false only when a fresh state snapshot provably excludes the topic.
// A false positive costs one wasted send; a false negative would lose a message and cannot
// happen for topics a peer has reported (Bloom filters have no false negatives).
func (c *meshCluster) mayNeed(peer NodeID, topic string) bool {
c.statesMu.Lock()
defer c.statesMu.Unlock()
state, ok := c.states[peer]
if !ok || time.Since(state.updatedAt) > 3*c.conf.StateInterval {
return true // No knowledge, or too old to trust for skipping
}
return state.topics.Contains(topic)
}
// peerWorker delivers batches of queued fan-out messages to a single peer. Batches form in the
// peer's LingerQueue (up to BatchLinger delay, flushed early on size/count caps); the worker
// exits when the queue is closed (peer left the registry, or mesh shutdown) and drained.
func (c *meshCluster) peerWorker(nodeID NodeID, q *peerQueue) {
defer c.wg.Done()
for frags := range q.queue.Dequeue() {
body := assembleMessageBody(frags)
log.Tag(tag).Debug("Sending batch of %d message(s) (%d bytes) to peer %s", len(frags), len(body), nodeID)
c.postToPeer(nodeID, messageURL(q.advertiseURL), contentTypeNDJSON, body)
metrics.ClusterBatchesSent.Inc()
}
}
// postToPeer POSTs a peer API payload, authenticated with the shared cluster secret. Failures
// are logged and counted, never retried: peer traffic is best-effort by design (messages are
// recovered via since= replay, state via the next periodic push).
func (c *meshCluster) postToPeer(nodeID NodeID, url, contentType string, payload []byte) {
req, err := http.NewRequestWithContext(c.ctx, http.MethodPost, url, bytes.NewReader(payload))
if err != nil {
metrics.ClusterSendErrors.Inc()
log.Tag(tag).Err(err).Warn("Failed to build request for peer %s", nodeID)
return
}
req.Header.Set("Content-Type", contentType)
req.Header.Set(secretHeader, c.conf.Secret)
req.Header.Set(originHeader, string(c.conf.NodeID))
resp, err := c.httpClient.Do(req)
if err != nil {
if c.ctx.Err() == nil {
metrics.ClusterSendErrors.Inc()
log.Tag(tag).Err(err).Warn("Failed to send to peer %s (%s)", nodeID, url)
}
return
}
resp.Body.Close()
if resp.StatusCode != http.StatusOK {
metrics.ClusterSendErrors.Inc()
log.Tag(tag).Warn("Peer %s (%s) rejected request with HTTP %d", nodeID, url, resp.StatusCode)
}
}
// handleMessage receives a batch of peer messages (NDJSON) and streams them to local
// subscribers line by line, delivering each message as it is decoded.
func (c *meshCluster) handleMessage(origin NodeID, w http.ResponseWriter, r *http.Request) {
// A batch can exceed its byte cap by one message, plus framing overhead
maxBodyBytes := int64(batchMaxBytes) + c.conf.MaxMessageBytes + 1024
received := 0
deliver := func(m *model.Message) {
received++
if ev := log.Tag(tag); ev.IsTrace() {
ev.Trace("Delivering message %s (topic %s) from peer %s", m.ID, m.Topic, origin)
}
c.deliver(m)
}
if err := decodeMessageBody(io.LimitReader(r.Body, maxBodyBytes), int(c.conf.MaxMessageBytes), deliver); err != nil {
w.WriteHeader(http.StatusBadRequest)
return
}
log.Tag(tag).Debug("Received batch of %d message(s) from peer %s", received, origin)
w.WriteHeader(http.StatusOK)
}
// handleState receives a peer's state envelope and applies each section it carries.
func (c *meshCluster) handleState(origin NodeID, w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(io.LimitReader(r.Body, stateMaxBytes))
if err != nil {
w.WriteHeader(http.StatusBadRequest)
return
}
var state apiState
if err := json.Unmarshal(body, &state); err != nil {
w.WriteHeader(http.StatusBadRequest)
return
}
if state.Topics != nil {
if err := c.applyTopicState(origin, state.Topics); err != nil {
w.WriteHeader(http.StatusBadRequest)
return
}
}
w.WriteHeader(http.StatusOK)
}
// applyTopicState updates what we know about a peer's subscriptions: a full snapshot replaces
// all prior knowledge, an incremental add merges into it. Increments without a baseline are
// ignored on purpose -- without a snapshot the peer is broadcast to anyway.
func (c *meshCluster) applyTopicState(origin NodeID, topics *apiStateTopics) error {
c.statesMu.Lock()
defer c.statesMu.Unlock()
if len(topics.Filter) > 0 {
filter, err := util.UnmarshalBloomFilter(topics.Filter)
if err != nil {
return err
}
c.states[origin] = &peerState{topics: filter, updatedAt: time.Now()}
log.Tag(tag).Debug("Received subscription state from peer %s (%d filter bytes)", origin, len(topics.Filter))
return nil
}
if state, ok := c.states[origin]; ok {
for _, topic := range topics.Added {
state.topics.Add(topic)
}
state.updatedAt = time.Now()
log.Tag(tag).Debug("Received %d announced topic(s) from peer %s", len(topics.Added), origin)
}
return nil
}
// pushState sends a full state snapshot to every live peer: a Bloom filter over the topics that
// currently have local subscribers. Sent directly (not via the linger queues -- state must not
// wait behind message batches); a lost push self-heals at the next interval. Topics without
// subscribers disappear simply by not being in the next snapshot.
func (c *meshCluster) pushState(peers []*registry.Peer) {
if len(peers) == 0 {
return
}
topics := c.topics()
filter := util.NewBloomFilter(len(topics), stateFilterFPRate)
for _, topic := range topics {
filter.Add(topic)
}
data, err := filter.MarshalBinary()
if err != nil {
return
}
body, err := json.Marshal(&apiState{Topics: &apiStateTopics{Filter: data}})
if err != nil {
return
}
log.Tag(tag).Debug("Pushing subscription state (%d topics, %d bytes) to %d peer(s)", len(topics), len(body), len(peers))
for _, p := range peers {
go c.postToPeer(NodeID(p.NodeID), stateURL(p.AdvertiseURL), contentTypeJSON, body)
}
metrics.ClusterStatePushes.Inc()
}
// BroadcastState immediately tells all live peers that these topics gained their first local
// subscriber, shrinking the window in which a publisher could wrongly skip this node from a
// full state interval down to about one round trip.
func (c *meshCluster) BroadcastState(state *State) {
if len(state.AddedTopics) == 0 {
return
}
peers, err := c.registry.Peers()
if err != nil || len(peers) == 0 {
return
}
body, err := json.Marshal(&apiState{Topics: &apiStateTopics{Added: state.AddedTopics}})
if err != nil {
return
}
log.Tag(tag).Debug("Broadcasting state (%d new topics) to %d peer(s)", len(state.AddedTopics), len(peers))
for _, p := range peers {
go c.postToPeer(NodeID(p.NodeID), stateURL(p.AdvertiseURL), contentTypeJSON, body)
}
}
// IsLeader reports whether this node currently holds singleton-job leadership.
func (c *meshCluster) IsLeader() bool {
return c.leader.IsLeader()
}
// Healthy reports whether this node's registry heartbeat is fresh enough that peers still
// forward messages to it (see the Cluster interface for the checker's fail-open duty).
func (c *meshCluster) Healthy() bool {
c.mu.Lock()
defer c.mu.Unlock()
return time.Since(c.lastRegistered) < c.conf.NodeTTL
}
// Close stops the mesh: it deregisters this node, releases leadership, stops all peer workers,
// and waits for them to exit.
func (c *meshCluster) Close() error {
c.cancel() // Stops the heartbeat loop and aborts in-flight peer deliveries
// Close the peer queues so their workers flush and exit; final sends are best-effort since
// the context is already canceled (parity with fire-and-forget delivery)
c.mu.Lock()
c.closed = true
for nodeID, q := range c.queues {
q.queue.Close()
delete(c.queues, nodeID)
}
c.mu.Unlock()
// Wait for the loops BEFORE deregistering: an in-flight heartbeat's Register would otherwise
// re-insert our row right after Deregister deleted it
c.wg.Wait()
if err := c.registry.Deregister(); err != nil {
log.Tag(tag).Err(err).Warn("Failed to deregister node")
}
c.leader.Close()
metrics.ClusterLeader.Set(0)
return nil
}
+590
View File
@@ -0,0 +1,590 @@
package cluster
import (
"bytes"
"encoding/json"
"fmt"
"io"
"net/http"
"net/http/httptest"
"strings"
"sync"
"testing"
"time"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/cluster/registry"
"heckel.io/ntfy/v2/db"
"heckel.io/ntfy/v2/db/pg"
dbtest "heckel.io/ntfy/v2/db/test"
"heckel.io/ntfy/v2/model"
"heckel.io/ntfy/v2/util"
)
const (
testSecret = "s3cret"
)
// openTestPool opens a dedicated connection pool to the given test schema, so that each simulated
// node has its own pool like real nodes would.
func openTestPool(t testing.TB, dsn string) *db.DB {
host, err := pg.Open(dsn)
require.Nil(t, err)
d := db.New(host, nil)
t.Cleanup(func() { d.Close() })
return d
}
func newTestMeshConfig(nodeID, advertiseURL string) *Config {
return &Config{
Enabled: true,
NodeID: NodeID(nodeID),
AdvertiseURL: advertiseURL,
Secret: testSecret,
HeartbeatInterval: 100 * time.Millisecond,
LeaderRenewInterval: 20 * time.Millisecond, // Lease duration 60ms, hold-off 120ms; keeps leadership tests fast
NodeTTL: time.Second, // Also the peer cache bound; short so fake peers registered mid-test are seen quickly
MaxMessageBytes: 1 << 20,
StateInterval: time.Minute, // Individual tests lower this to exercise state pushes
}
}
// registerFakePeer registers a fake peer via the registry (creating the table if the mesh has
// not been constructed yet): tests register fakes before the mesh boots, since its first
// heartbeat caches the peer list. The fake never refreshes its heartbeat.
func registerFakePeer(t testing.TB, pool *db.DB, nodeID NodeID, url string) {
t.Helper()
reg, err := registry.New(pool, string(nodeID), url, time.Minute)
require.Nil(t, err)
require.Nil(t, reg.Register())
}
func waitFor(t *testing.T, f func() bool) {
t.Helper()
for i := 0; i < 100; i++ {
if f() {
return
}
time.Sleep(50 * time.Millisecond)
}
t.Fatal("timed out waiting for condition")
}
func TestMesh_CrossNodeDelivery(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
poolA, poolB := openTestPool(t, schemaDSN), openTestPool(t, schemaDSN)
var mu sync.Mutex
var received []*model.Message
var meshB *meshCluster
srvB := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
meshB.ServeHTTP(w, r)
}))
defer srvB.Close()
meshB, err := newMeshCluster(newTestMeshConfig("node-b", srvB.URL), poolB, func(m *model.Message) {
mu.Lock()
defer mu.Unlock()
received = append(received, m)
}, nil)
require.Nil(t, err)
defer meshB.Close()
meshA, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), poolA, func(m *model.Message) {
t.Error("node A must not receive its own relayed message")
}, nil)
require.Nil(t, err)
defer meshA.Close()
msg := model.NewDefaultMessage("mytopic", "hello cross-node")
require.Nil(t, meshA.ForwardMessage(msg))
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return len(received) == 1
})
mu.Lock()
defer mu.Unlock()
require.Equal(t, "mytopic", received[0].Topic)
require.Equal(t, "hello cross-node", received[0].Message)
}
func TestMesh_PeerAPI_Auth(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var delivered int
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, func(m *model.Message) {
delivered++
}, nil)
require.Nil(t, err)
defer mesh.Close()
frag, err := marshalMessage(model.NewDefaultMessage("mytopic", "hi"))
require.Nil(t, err)
payload := assembleMessageBody([][]byte{frag})
// Wrong secret -> 401, not delivered
rr := httptest.NewRecorder()
req := httptest.NewRequest("POST", MessagePath, strings.NewReader(string(payload)))
req.Header.Set(secretHeader, "wrong")
req.Header.Set(originHeader, "node-b")
mesh.ServeHTTP(rr, req)
require.Equal(t, 401, rr.Code)
// Missing secret -> 401, not delivered
rr = httptest.NewRecorder()
mesh.ServeHTTP(rr, httptest.NewRequest("POST", MessagePath, strings.NewReader(string(payload))))
require.Equal(t, 401, rr.Code)
require.Equal(t, 0, delivered)
// Missing origin -> 400, not delivered
rr = httptest.NewRecorder()
req = httptest.NewRequest("POST", MessagePath, strings.NewReader(string(payload)))
req.Header.Set(secretHeader, testSecret)
mesh.ServeHTTP(rr, req)
require.Equal(t, 400, rr.Code)
require.Equal(t, 0, delivered)
// Correct secret and origin -> 200, delivered
rr = httptest.NewRecorder()
req = httptest.NewRequest("POST", MessagePath, strings.NewReader(string(payload)))
req.Header.Set(secretHeader, testSecret)
req.Header.Set(originHeader, "node-b")
mesh.ServeHTTP(rr, req)
require.Equal(t, 200, rr.Code)
require.Equal(t, 1, delivered)
}
func TestMesh_PeerAPI_SelfOrigin(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var delivered int
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, func(m *model.Message) {
delivered++
}, nil)
require.Nil(t, err)
defer mesh.Close()
// A request that carries this node's own broadcasts must not be re-delivered (loop prevention)
frag, err := marshalMessage(model.NewDefaultMessage("mytopic", "loop"))
require.Nil(t, err)
payload := assembleMessageBody([][]byte{frag})
rr := httptest.NewRecorder()
req := httptest.NewRequest("POST", MessagePath, strings.NewReader(string(payload)))
req.Header.Set(secretHeader, testSecret)
req.Header.Set(originHeader, "node-a") // Same as the receiving node's ID
mesh.ServeHTTP(rr, req)
require.Equal(t, 200, rr.Code)
require.Equal(t, 0, delivered)
}
func TestMesh_SlowPeerIsolation(t *testing.T) {
// A wedged peer must not delay delivery to healthy peers: each peer has its own queue and
// delivery worker. With a shared send queue (the design this replaces), the slow peer's
// requests would occupy all delivery workers and starve the fast peer.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var mu sync.Mutex
fastReceived := 0 // Messages, not requests: with batching, one request can carry many
srvFast := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(r.Body)
require.Nil(t, err)
messages, err := unmarshalMessageBody(body, 1<<20)
require.Nil(t, err)
mu.Lock()
fastReceived += len(messages)
mu.Unlock()
w.WriteHeader(http.StatusOK)
}))
defer srvFast.Close()
release := make(chan struct{})
srvSlow := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
<-release // Wedged until the end of the test
w.WriteHeader(http.StatusOK)
}))
defer srvSlow.Close()
defer close(release)
// Register the fake peers before the mesh boots; its first heartbeat caches the peer list
for i, url := range []string{srvFast.URL, srvSlow.URL} {
registerFakePeer(t, pool, NodeID(fmt.Sprintf("node-fake-%d", i)), url)
}
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
const n = 20
for i := 0; i < n; i++ {
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("mytopic", fmt.Sprintf("message %d", i))))
}
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return fastReceived == n
})
}
func TestMesh_BatchCoalescing(t *testing.T) {
// Messages published within the linger window arrive as batches: fewer HTTP requests than
// messages, with nothing lost. Fails against a one-request-per-message sender.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var mu sync.Mutex
requests, messages := 0, 0
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(r.Body)
require.Nil(t, err)
decoded, err := unmarshalMessageBody(body, 1<<20)
require.Nil(t, err)
mu.Lock()
requests++
messages += len(decoded)
mu.Unlock()
w.WriteHeader(http.StatusOK)
}))
defer srv.Close()
registerFakePeer(t, pool, "node-fake", srv.URL)
conf := newTestMeshConfig("node-a", "http://127.0.0.1:1")
conf.BatchLinger = 150 * time.Millisecond
mesh, err := newMeshCluster(conf, pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
const n = 20
for i := 0; i < n; i++ {
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("mytopic", fmt.Sprintf("message %d", i))))
}
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return messages == n
})
mu.Lock()
defer mu.Unlock()
require.Less(t, requests, 5, "expected %d messages coalesced into few requests, got %d", n, requests)
}
func TestMesh_DeadPeerRemovedAndRejoin(t *testing.T) {
// A peer that dies ungracefully (no Deregister) stops refreshing its heartbeat: after the
// TTL it no longer counts as live (no more sends), its queue/worker are reconciled away, the
// leader prunes its registry row, and a re-registered peer starts receiving again.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var mu sync.Mutex
received := 0
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(r.Body)
require.Nil(t, err)
messages, err := unmarshalMessageBody(body, 1<<20)
require.Nil(t, err)
mu.Lock()
received += len(messages)
mu.Unlock()
w.WriteHeader(http.StatusOK)
}))
defer srv.Close()
conf := newTestMeshConfig("node-a", "http://127.0.0.1:1")
conf.NodeTTL = 300 * time.Millisecond // Fast expiry so the test observes TTL-based removal
mesh, err := newMeshCluster(conf, pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
// The fake peer registers once and then "dies": its heartbeat is never refreshed
registerFakePeer(t, pool, "node-dead", srv.URL)
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("mytopic", "while alive")))
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return received == 1
})
// After the TTL, the peer is no longer live: its queue is reconciled away and its registry
// row is pruned by the leader (this mesh is the only real node, so it holds the lock)
waitFor(t, func() bool {
mesh.mu.Lock()
defer mesh.mu.Unlock()
return len(mesh.queues) == 0
})
waitFor(t, func() bool {
var count int
require.Nil(t, pool.QueryRow(`SELECT COUNT(*) FROM node_registry WHERE node_id = 'node-dead'`).Scan(&count))
return count == 0
})
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("mytopic", "while dead")))
time.Sleep(250 * time.Millisecond) // Give a wrong implementation time to deliver anyway
mu.Lock()
require.Equal(t, 1, received) // Only the first message arrived
mu.Unlock()
// The peer comes back (same node ID, fresh heartbeat) and receives messages again; the
// relay retries because the peer list is cached for up to the node TTL
registerFakePeer(t, pool, "node-dead", srv.URL)
waitFor(t, func() bool {
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("mytopic", "after rejoin")))
mu.Lock()
defer mu.Unlock()
return received > 1
})
}
func TestMesh_ForwardAfterClose(t *testing.T) {
// A ForwardMessage racing shutdown (e.g. an in-flight publish during server Stop) must not spawn
// a new peer queue and worker after Close: the worker would never exit (its queue is never
// closed) and nothing waits for it.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, nil, nil)
require.Nil(t, err)
registerFakePeer(t, pool, "node-peer", "http://127.0.0.1:1")
require.Nil(t, mesh.Close())
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("mytopic", "too late"))) // Dropped silently
mesh.mu.Lock()
defer mesh.mu.Unlock()
require.Empty(t, mesh.queues)
}
func TestMesh_LeaderFailover(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
poolA, poolB := openTestPool(t, schemaDSN), openTestPool(t, schemaDSN)
meshA, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), poolA, nil, nil)
require.Nil(t, err)
defer meshA.Close()
meshB, err := newMeshCluster(newTestMeshConfig("node-b", "http://127.0.0.1:1"), poolB, nil, nil)
require.Nil(t, err)
defer meshB.Close()
// Exactly one node becomes leader
waitFor(t, func() bool {
return meshA.IsLeader() != meshB.IsLeader() // Exactly one
})
// The leader steps down; the follower takes over
leader, follower := meshA, meshB
if meshB.IsLeader() {
leader, follower = meshB, meshA
}
require.Nil(t, leader.Close())
waitFor(t, follower.IsLeader)
}
func TestMesh_CloseDeregisters(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, nil, nil)
require.Nil(t, err)
var count int
require.Nil(t, pool.QueryRow(`SELECT COUNT(*) FROM node_registry WHERE node_id = 'node-a'`).Scan(&count))
require.Equal(t, 1, count)
require.Nil(t, mesh.Close())
require.Nil(t, pool.QueryRow(`SELECT COUNT(*) FROM node_registry WHERE node_id = 'node-a'`).Scan(&count))
require.Equal(t, 0, count)
}
// postState delivers a state envelope to a mesh's peer API, as a peer would.
func postState(c *meshCluster, origin NodeID, state *apiState) *httptest.ResponseRecorder {
body, err := json.Marshal(state)
if err != nil {
panic(err)
}
rr := httptest.NewRecorder()
req := httptest.NewRequest("POST", StatePath, bytes.NewReader(body))
req.Header.Set(secretHeader, testSecret)
req.Header.Set(originHeader, string(origin))
c.ServeHTTP(rr, req)
return rr
}
// topicFilter builds a marshaled Bloom filter over the given topics.
func topicFilter(t *testing.T, topics ...string) []byte {
t.Helper()
filter := util.NewBloomFilter(len(topics), 0.01)
for _, topic := range topics {
filter.Add(topic)
}
data, err := filter.MarshalBinary()
require.Nil(t, err)
return data
}
func TestMesh_RouteSkipsUnsubscribedPeer(t *testing.T) {
// A peer whose fresh state provably excludes a topic is not contacted for it; a topic in its
// state is delivered as usual.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var mu sync.Mutex
received := 0
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, err := io.ReadAll(r.Body)
require.Nil(t, err)
messages, err := unmarshalMessageBody(body, 1<<20)
require.Nil(t, err)
mu.Lock()
received += len(messages)
mu.Unlock()
w.WriteHeader(http.StatusOK)
}))
defer srv.Close()
registerFakePeer(t, pool, "node-b", srv.URL)
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
// node-b reports subscribers only for "subscribed-topic"
rr := postState(mesh, "node-b", &apiState{Topics: &apiStateTopics{Filter: topicFilter(t, "subscribed-topic")}})
require.Equal(t, 200, rr.Code)
// A topic outside the peer's state is skipped
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("other-topic", "skipped")))
time.Sleep(300 * time.Millisecond) // Give a wrong implementation time to deliver anyway
mu.Lock()
require.Equal(t, 0, received)
mu.Unlock()
// A topic inside the peer's state is delivered
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("subscribed-topic", "delivered")))
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return received == 1
})
}
func TestMesh_RouteBroadcastsOnStaleState(t *testing.T) {
// State too old to trust cannot justify skipping: the peer is broadcast to as if unknown.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
var mu sync.Mutex
received := 0
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.URL.Path == MessagePath { // The mesh also pushes state here; count only messages
mu.Lock()
received++
mu.Unlock()
}
w.WriteHeader(http.StatusOK)
}))
defer srv.Close()
registerFakePeer(t, pool, "node-b", srv.URL)
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
rr := postState(mesh, "node-b", &apiState{Topics: &apiStateTopics{Filter: topicFilter(t, "subscribed-topic")}})
require.Equal(t, 200, rr.Code)
// Age the state beyond the trust window
mesh.statesMu.Lock()
mesh.states["node-b"].updatedAt = time.Now().Add(-time.Hour)
mesh.statesMu.Unlock()
require.Nil(t, mesh.ForwardMessage(model.NewDefaultMessage("other-topic", "broadcast anyway")))
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return received == 1
})
}
func TestMesh_StatePushReplacesAndRemoves(t *testing.T) {
// Node A periodically pushes a full snapshot of its live topics to node B; each snapshot
// REPLACES B's knowledge, so topics that lost their subscribers disappear without any
// explicit removal protocol.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
poolA, poolB := openTestPool(t, schemaDSN), openTestPool(t, schemaDSN)
var topicsMu sync.Mutex
topicsA := []string{"topic-1"}
source := func() []string {
topicsMu.Lock()
defer topicsMu.Unlock()
return append([]string{}, topicsA...)
}
var meshB *meshCluster
srvB := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
meshB.ServeHTTP(w, r)
}))
defer srvB.Close()
meshB, err := newMeshCluster(newTestMeshConfig("node-b", srvB.URL), poolB, nil, nil)
require.Nil(t, err)
defer meshB.Close()
confA := newTestMeshConfig("node-a", "http://127.0.0.1:1")
confA.StateInterval = 200 * time.Millisecond
meshA, err := newMeshCluster(confA, poolA, nil, source)
require.Nil(t, err)
defer meshA.Close()
// B learns A's topics via the periodic push
knows := func(topic string) func() bool {
return func() bool {
meshB.statesMu.Lock()
defer meshB.statesMu.Unlock()
state, ok := meshB.states["node-a"]
return ok && state.topics.Contains(topic)
}
}
waitFor(t, knows("topic-1"))
// A's subscribers change; the next snapshot replaces the old knowledge entirely
topicsMu.Lock()
topicsA = []string{"topic-2"}
topicsMu.Unlock()
waitFor(t, knows("topic-2"))
waitFor(t, func() bool { return !knows("topic-1")() })
}
func TestMesh_AnnounceClosesWindow(t *testing.T) {
// A topic gaining its first subscriber is announced immediately, so peers learn about it
// without waiting for the next full state push.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
poolA, poolB := openTestPool(t, schemaDSN), openTestPool(t, schemaDSN)
var meshB *meshCluster
srvB := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
meshB.ServeHTTP(w, r)
}))
defer srvB.Close()
meshB, err := newMeshCluster(newTestMeshConfig("node-b", srvB.URL), poolB, nil, nil)
require.Nil(t, err)
defer meshB.Close()
confA := newTestMeshConfig("node-a", "http://127.0.0.1:1")
confA.StateInterval = 200 * time.Millisecond // One full push establishes the baseline
meshA, err := newMeshCluster(confA, poolA, nil, func() []string { return []string{"existing"} })
require.Nil(t, err)
defer meshA.Close()
waitFor(t, func() bool {
meshB.statesMu.Lock()
defer meshB.statesMu.Unlock()
_, ok := meshB.states["node-a"]
return ok
})
// Announcements merge into the baseline right away
meshA.BroadcastState(&State{AddedTopics: []string{"fresh-topic"}})
waitFor(t, func() bool {
meshB.statesMu.Lock()
defer meshB.statesMu.Unlock()
state, ok := meshB.states["node-a"]
return ok && state.topics.Contains("fresh-topic")
})
}
func TestMesh_StateOfDepartedPeerPruned(t *testing.T) {
// peerState is push-driven and can arrive before the peer is visible in the registry, so it
// must survive reconcile while fresh -- but a departed peer's state must not leak forever:
// once it is both absent from the registry and stale past the trust window, it is pruned.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
rr := postState(mesh, "node-gone", &apiState{Topics: &apiStateTopics{Filter: topicFilter(t, "some-topic")}})
require.Equal(t, 200, rr.Code)
// Fresh state of an unknown peer survives reconcile (the new-node visibility window)
mesh.reconcilePeers(nil)
mesh.statesMu.Lock()
_, ok := mesh.states["node-gone"]
mesh.statesMu.Unlock()
require.True(t, ok)
// Stale state of an absent peer is pruned
mesh.statesMu.Lock()
mesh.states["node-gone"].updatedAt = time.Now().Add(-time.Hour)
mesh.statesMu.Unlock()
mesh.reconcilePeers(nil)
mesh.statesMu.Lock()
_, ok = mesh.states["node-gone"]
mesh.statesMu.Unlock()
require.False(t, ok)
}
func TestMesh_HealthyReflectsRegistration(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
mesh, err := newMeshCluster(newTestMeshConfig("node-a", "http://127.0.0.1:1"), pool, nil, nil)
require.Nil(t, err)
defer mesh.Close()
require.True(t, mesh.Healthy()) // Registered synchronously at construction
// Stale heartbeat: peers stop forwarding to this node, so it must report unhealthy
mesh.mu.Lock()
mesh.lastRegistered = time.Now().Add(-2 * mesh.conf.NodeTTL)
mesh.mu.Unlock()
require.False(t, mesh.Healthy())
// A successful heartbeat restores health
require.Nil(t, mesh.heartbeat())
require.True(t, mesh.Healthy())
}
+26
View File
@@ -0,0 +1,26 @@
package cluster
import (
"net/http"
"heckel.io/ntfy/v2/model"
)
// nopCluster is the single-node default: it drops all relayed messages, rejects peer API requests, and
// reports this node as leader (a single node is trivially the leader, so leader-gated jobs need
// no special-casing in single-node mode).
type nopCluster struct{}
func (c *nopCluster) ForwardMessage(_ *model.Message) error { return nil }
func (c *nopCluster) ServeHTTP(w http.ResponseWriter, _ *http.Request) {
w.WriteHeader(http.StatusNotFound)
}
func (c *nopCluster) BroadcastState(_ *State) {}
func (c *nopCluster) IsLeader() bool { return true }
func (c *nopCluster) Healthy() bool { return true }
func (c *nopCluster) Close() error { return nil }
+84
View File
@@ -0,0 +1,84 @@
package cluster
import (
"bytes"
"net/http/httptest"
"net/netip"
"testing"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/model"
)
func TestDeliver_RoundTrip(t *testing.T) {
// The fan-out body is NDJSON: one apiDeliverMessage per line, joined from pre-marshaled
// fragments; the origin travels in a header, not the body
m1 := model.NewDefaultMessage("mytopic", "my message")
m1.Sender = netip.MustParseAddr("1.2.3.4")
m1.User = "u_abc"
m2 := model.NewDefaultMessage("othertopic", "other message")
frag1, err := marshalMessage(m1)
require.Nil(t, err)
frag2, err := marshalMessage(m2)
require.Nil(t, err)
messages, err := unmarshalMessageBody(assembleMessageBody([][]byte{frag1, frag2}), 1<<20)
require.Nil(t, err)
require.Len(t, messages, 2)
require.Equal(t, "mytopic", messages[0].Topic)
require.Equal(t, "my message", messages[0].Message)
// Sender and User are json:"-" on model.Message; the lines must carry and reattach them
require.Equal(t, netip.MustParseAddr("1.2.3.4"), messages[0].Sender)
require.Equal(t, "u_abc", messages[0].User)
require.Equal(t, "othertopic", messages[1].Topic)
require.False(t, messages[1].Sender.IsValid())
}
func TestDeliver_SingleMessage(t *testing.T) {
// A single message is just a one-line body; there is no separate single-message format
frag, err := marshalMessage(model.NewDefaultMessage("mytopic", "hi"))
require.Nil(t, err)
messages, err := unmarshalMessageBody(assembleMessageBody([][]byte{frag}), 1<<20)
require.Nil(t, err)
require.Len(t, messages, 1)
}
func TestDeliver_MalformedLinesSkipped(t *testing.T) {
// Fan-out is fire-and-forget: a malformed or message-less line is skipped (and logged), the
// remaining lines are still delivered
frag, err := marshalMessage(model.NewDefaultMessage("mytopic", "good"))
require.Nil(t, err)
body := []byte("this is not json\n{\"sender\":\"1.2.3.4\"}\n" + string(frag) + "\n\n")
messages, err := unmarshalMessageBody(body, 1<<20)
require.Nil(t, err)
require.Len(t, messages, 1)
require.Equal(t, "good", messages[0].Message)
}
// unmarshalMessageBody is a test helper collecting the messages of an NDJSON message body.
func unmarshalMessageBody(body []byte, maxLineBytes int) ([]*model.Message, error) {
var messages []*model.Message
err := decodeMessageBody(bytes.NewReader(body), maxLineBytes, func(m *model.Message) {
messages = append(messages, m)
})
return messages, err
}
func TestNop(t *testing.T) {
b, err := New(&Config{}, nil, nil, nil) // not enabled -> nop cluster, no database required
require.Nil(t, err)
require.IsType(t, &nopCluster{}, b)
require.Nil(t, b.ForwardMessage(model.NewDefaultMessage("mytopic", "hi")))
// A single node is trivially the leader, so leader-gated jobs run without special-casing
require.True(t, b.IsLeader())
require.True(t, b.Healthy())
rr := httptest.NewRecorder()
b.ServeHTTP(rr, httptest.NewRequest("POST", MessagePath, nil))
require.Equal(t, 404, rr.Code)
require.Nil(t, b.Close())
}
func TestNew_EnabledRequiresDatabase(t *testing.T) {
_, err := New(&Config{Enabled: true, Secret: "secret"}, nil, nil, nil)
require.Error(t, err)
require.Contains(t, err.Error(), "database")
}
+149
View File
@@ -0,0 +1,149 @@
// Package registry implements cluster membership: each node upserts its own row into the
// node_registry table with a fresh heartbeat, and discovers its peers by reading the other
// fresh rows. Node IDs are plain strings here; the cluster package layers its NodeID type on
// top.
package registry
import (
"sync"
"time"
"heckel.io/ntfy/v2/db"
"heckel.io/ntfy/v2/db/schema"
)
// Registry queries
const (
upsertNodeQuery = `
INSERT INTO node_registry (node_id, advertise_url, last_heartbeat)
VALUES ($1, $2, $3)
ON CONFLICT (node_id) DO UPDATE SET advertise_url = EXCLUDED.advertise_url, last_heartbeat = EXCLUDED.last_heartbeat
`
selectPeersQuery = `SELECT node_id, advertise_url FROM node_registry WHERE last_heartbeat >= $1 AND node_id != $2`
pruneStaleNodesQuery = `DELETE FROM node_registry WHERE last_heartbeat < $1`
deleteNodeQuery = `DELETE FROM node_registry WHERE node_id = $1`
)
// Schema version and queries
const (
schemaVersion = 1
schemaStoreKey = "node_registry"
)
var (
createTable = schema.AsMigrateFunc(`
CREATE TABLE IF NOT EXISTS node_registry (
node_id TEXT PRIMARY KEY,
advertise_url TEXT NOT NULL,
last_heartbeat BIGINT NOT NULL
)
`)
)
// Peer is a live remote node as read from the registry.
type Peer struct {
NodeID string
AdvertiseURL string
}
// Registry is the node membership table (control plane): each node upserts its own row with a
// fresh heartbeat every few seconds, and peers are the other rows with a heartbeat newer than
// the TTL. Stale rows are pruned by the leader. The TTL bounds membership staleness in BOTH
// directions: how long a silent node still counts as live, and how long the cached peer list is
// served before a re-read -- so a new node may take up to a TTL to become visible.
type Registry struct {
pool *db.DB
nodeID string
advertiseURL string
ttl time.Duration
peers []*Peer // cached peer list
peersFetched time.Time
mu sync.Mutex // Protects peers and peersFetched
}
// New creates or migrates the registry schema and returns this node's membership handle. It
// does NOT register the node: joining the cluster is an explicit Register call, owned by the
// caller, so read-only uses of the registry stay side-effect free.
func New(pool *db.DB, nodeID, advertiseURL string, ttl time.Duration) (*Registry, error) {
if err := schema.Migrate(pool.Primary(), schema.Postgres, schemaStoreKey, schemaVersion, createTable, nil); err != nil {
return nil, err
}
return &Registry{
pool: pool,
nodeID: nodeID,
advertiseURL: advertiseURL,
ttl: ttl,
}, nil
}
// Register upserts this node into the registry with a fresh heartbeat. It is a pure write: it
// does not touch the peer cache, because our own row is excluded from Peers() anyway.
func (r *Registry) Register() error {
_, err := r.pool.Exec(upsertNodeQuery, r.nodeID, r.advertiseURL, time.Now().Unix())
return err
}
// Peers returns the current set of live peer nodes (all registry rows with a fresh heartbeat,
// excluding this node), cached for the TTL.
func (r *Registry) Peers() ([]*Peer, error) {
r.mu.Lock()
if r.peers != nil && time.Since(r.peersFetched) < r.ttl {
peers := r.peers
r.mu.Unlock()
return peers, nil
}
r.mu.Unlock()
peers, err := r.queryPeers()
if err != nil {
// Serve the last-known peer list during database hiccups: fan-out keeps flowing to
// known peers instead of erroring (and logging) once per published message for the
// duration of the outage. Dead peers in the stale list only cost failed sends.
r.mu.Lock()
defer r.mu.Unlock()
if r.peers != nil {
return r.peers, nil
}
return nil, err
}
r.mu.Lock()
r.peers = peers
r.peersFetched = time.Now()
r.mu.Unlock()
return peers, nil
}
// Prune deletes registry rows whose heartbeat is long expired. Only the leader calls this; the
// grace period of 3x the TTL avoids deleting rows of nodes that are merely slow to heartbeat.
func (r *Registry) Prune() error {
_, err := r.pool.Exec(pruneStaleNodesQuery, time.Now().Add(-3*r.ttl).Unix())
return err
}
// Deregister deletes this node's registry row; called on shutdown.
func (r *Registry) Deregister() error {
_, err := r.pool.Exec(deleteNodeQuery, r.nodeID)
return err
}
// queryPeers reads the current live peer set from the registry table.
func (r *Registry) queryPeers() ([]*Peer, error) {
cutoff := time.Now().Add(-r.ttl).Unix()
rows, err := r.pool.Query(selectPeersQuery, cutoff, r.nodeID)
if err != nil {
return nil, err
}
defer rows.Close()
peers := make([]*Peer, 0)
for rows.Next() {
p := &Peer{}
if err := rows.Scan(&p.NodeID, &p.AdvertiseURL); err != nil {
return nil, err
}
peers = append(peers, p)
}
if err := rows.Err(); err != nil {
return nil, err
}
return peers, nil
}
+222
View File
@@ -0,0 +1,222 @@
package registry
import (
"fmt"
"testing"
"time"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/db"
"heckel.io/ntfy/v2/db/pg"
dbtest "heckel.io/ntfy/v2/db/test"
)
func openTestPool(t *testing.T, dsn string) *db.DB {
t.Helper()
host, err := pg.Open(dsn)
require.Nil(t, err)
d := db.New(host, nil)
t.Cleanup(func() { d.Close() })
return d
}
func TestRegistry_NewDoesNotRegister(t *testing.T) {
// New only sets up the schema and the identity handle; joining the cluster is an explicit
// Register call, owned by the caller (the mesh registers synchronously at construction).
// This keeps read-only uses (ops tooling, future admin endpoints) side-effect free.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
require.Equal(t, 0, countRows(t, pool, "node-1"))
require.Nil(t, r1.Register())
require.Equal(t, 1, countRows(t, pool, "node-1"))
}
func TestRegistry_RegisterAndPeers(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
require.Nil(t, r1.Register())
r2, err := New(pool, "node-2", "http://10.0.0.2:2587", time.Minute)
require.Nil(t, err)
require.Nil(t, r2.Register())
// Each node sees the other, never itself
peers, err := r1.Peers()
require.Nil(t, err)
require.Len(t, peers, 1)
require.Equal(t, "node-2", peers[0].NodeID)
require.Equal(t, "http://10.0.0.2:2587", peers[0].AdvertiseURL)
peers, err = r2.Peers()
require.Nil(t, err)
require.Len(t, peers, 1)
require.Equal(t, "node-1", peers[0].NodeID)
}
func TestRegistry_ReRegisterUpdatesAdvertiseURL(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
// The same node comes back under a new address; the upsert replaces the row
old, err := New(pool, "node-2", "http://old:2587", time.Minute)
require.Nil(t, err)
require.Nil(t, old.Register())
renewed, err := New(pool, "node-2", "http://new:2587", time.Minute)
require.Nil(t, err)
require.Nil(t, renewed.Register())
expireCache(r1)
peers, err := r1.Peers()
require.Nil(t, err)
require.Len(t, peers, 1)
require.Equal(t, "http://new:2587", peers[0].AdvertiseURL)
}
func TestRegistry_PeersCachedForTTL(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
peers, err := r1.Peers()
require.Nil(t, err)
require.Empty(t, peers)
// A node joining after the cache was populated is invisible until the cache expires
r2, err := New(pool, "node-2", "http://10.0.0.2:2587", time.Minute)
require.Nil(t, err)
require.Nil(t, r2.Register())
peers, err = r1.Peers()
require.Nil(t, err)
require.Empty(t, peers)
expireCache(r1)
peers, err = r1.Peers()
require.Nil(t, err)
require.Len(t, peers, 1)
}
func TestRegistry_TTLExcludesSilentNodes(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
// A node whose heartbeat is older than the TTL does not count as live
_, err = pool.Exec(upsertNodeQuery, "node-silent", "http://10.0.0.9:2587", time.Now().Add(-2*time.Minute).Unix())
require.Nil(t, err)
peers, err := r1.Peers()
require.Nil(t, err)
require.Empty(t, peers)
}
func TestRegistry_PruneDeletesLongDeadOnly(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
// One node beyond the 3x TTL grace period, one merely stale
_, err = pool.Exec(upsertNodeQuery, "node-long-dead", "http://10.0.0.8:2587", time.Now().Add(-4*time.Minute).Unix())
require.Nil(t, err)
_, err = pool.Exec(upsertNodeQuery, "node-slow", "http://10.0.0.9:2587", time.Now().Add(-2*time.Minute).Unix())
require.Nil(t, err)
require.Nil(t, r1.Prune())
require.Equal(t, 0, countRows(t, pool, "node-long-dead"))
require.Equal(t, 1, countRows(t, pool, "node-slow")) // Slow, not dead: kept
}
func TestRegistry_Deregister(t *testing.T) {
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
require.Nil(t, r1.Register())
require.Equal(t, 1, countRows(t, pool, "node-1"))
require.Nil(t, r1.Deregister())
require.Equal(t, 0, countRows(t, pool, "node-1"))
}
func TestRegistry_PeersStaleCacheOnError(t *testing.T) {
// During a database hiccup, Peers serves the last-known peer list instead of erroring:
// fan-out keeps flowing to known peers, and the publish path does not log a warning per
// message for the duration of the outage.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
r1, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
r2, err := New(pool, "node-2", "http://10.0.0.2:2587", time.Minute)
require.Nil(t, err)
require.Nil(t, r2.Register())
peers, err := r1.Peers()
require.Nil(t, err)
require.Len(t, peers, 1)
// Expire the cache and break the database; the stale list must still be served
expireCache(r1)
require.Nil(t, pool.Close())
peers, err = r1.Peers()
require.Nil(t, err)
require.Len(t, peers, 1)
require.Equal(t, "node-2", peers[0].NodeID)
}
func TestRegistry_ConcurrentCreate(t *testing.T) {
// Multiple nodes cold-booting on a fresh database must not race on table creation: CREATE
// TABLE IF NOT EXISTS is not atomic in PostgreSQL, so creation is serialized via an advisory
// lock. Without it, this test fails sporadically with a duplicate-key error on pg_class.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
const n = 8
errs := make(chan error, n)
for i := 0; i < n; i++ {
go func(i int) {
pool, err := pg.Open(schemaDSN)
if err != nil {
errs <- err
return
}
defer pool.DB.Close()
_, err = New(db.New(pool, nil), fmt.Sprintf("node-%d", i), "http://127.0.0.1:1", time.Second)
errs <- err
}(i)
}
for i := 0; i < n; i++ {
require.Nil(t, <-errs)
}
}
func TestRegistry_SchemaVersionWritten(t *testing.T) {
// The registry participates in the shared schema_version framework like every other store,
// so future table changes can be applied as migrations.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
_, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
var version int
require.Nil(t, pool.QueryRow(`SELECT version FROM schema_version WHERE store = $1`, schemaStoreKey).Scan(&version))
require.Equal(t, schemaVersion, version)
// Setup is idempotent: a second node boots against the migrated schema
_, err = New(pool, "node-2", "http://10.0.0.2:2587", time.Minute)
require.Nil(t, err)
}
func TestRegistry_SchemaVersionFromTheFuture(t *testing.T) {
// A node running older code must refuse to touch a schema migrated by newer code
schemaDSN := dbtest.CreateTestPostgresSchema(t)
pool := openTestPool(t, schemaDSN)
_, err := New(pool, "node-1", "http://10.0.0.1:2587", time.Minute)
require.Nil(t, err)
_, err = pool.Exec(`UPDATE schema_version SET version = 99 WHERE store = $1`, schemaStoreKey)
require.Nil(t, err)
_, err = New(pool, "node-2", "http://10.0.0.2:2587", time.Minute)
require.Error(t, err)
}
// expireCache forces the next Peers() call to re-read the registry table.
func expireCache(r *Registry) {
r.mu.Lock()
r.peersFetched = time.Time{}
r.mu.Unlock()
}
func countRows(t *testing.T, pool *db.DB, nodeID string) int {
t.Helper()
var count int
require.Nil(t, pool.QueryRow(`SELECT COUNT(*) FROM node_registry WHERE node_id = $1`, nodeID).Scan(&count))
return count
}
+76
View File
@@ -0,0 +1,76 @@
package cluster
import (
"time"
"heckel.io/ntfy/v2/model"
"heckel.io/ntfy/v2/util"
)
// Config configures the cluster. It is assembled by the server from its own config, which keeps
// this package free of server types.
type Config struct {
Enabled bool // Master switch; when false, New returns the nop cluster
NodeID NodeID // Stable per-node identifier; required
AdvertiseURL string // Base URL peers use to reach this node's fan-out endpoint
Secret string // Shared secret authenticating node-to-node fan-out requests
HeartbeatInterval time.Duration // How often the node registry heartbeat is refreshed
NodeTTL time.Duration // Registry rows older than this do not count as live peers
BatchLinger time.Duration // How long messages wait in a peer queue to form a batch; 0 = send immediately
StateInterval time.Duration // How often the full subscription state is pushed to peers
MaxMessageBytes int64 // Upper bound for a single message on the wire (batch limits derive from this)
LeaderRenewInterval time.Duration // Overrides the leader lease renewal cadence; tests only, 0 = default
}
// DeliverFunc hands a message received from a peer node to this node's local subscribers. The
// server supplies it, which inverts the dependency: this package never imports the server.
type DeliverFunc func(m *model.Message)
// State is a subscription-state delta for Cluster.BroadcastState.
type State struct {
AddedTopics []string // Topics that just gained their first local subscriber on this node
}
// TopicsFunc returns the topics that currently have at least one live subscriber, computed
// fresh on every call: membership is never tracked as a list, so topics "leave" simply by not
// appearing in the next snapshot. The server supplies it (same inversion as DeliverFunc).
type TopicsFunc func() []string
// apiMessage is one line of a message request body (NDJSON: one message per line; a single
// message is just a one-line body). It carries the two fields that model.Message does not
// serialize to JSON (Sender and User), which are needed to reconstruct the visitor on the
// receiving node. The origin node travels in a request header, not in the body.
type apiMessage struct {
Sender string `json:"sender,omitempty"`
User string `json:"user,omitempty"`
Message *model.Message `json:"message"`
}
// apiState is the peer state-exchange envelope. Each concern is an optional section; future
// concerns (rate limit counters, stats) become siblings of Topics.
type apiState struct {
Topics *apiStateTopics `json:"topics,omitempty"`
}
// apiStateTopics carries a peer's subscription knowledge: either a full snapshot (Filter, a
// marshaled Bloom filter over the topics with live subscribers) replacing all prior knowledge,
// or an incremental update (Added) merged into it.
type apiStateTopics struct {
Filter []byte `json:"filter,omitempty"`
Added []string `json:"added,omitempty"`
}
// peerState is what a peer last told us about itself; ForwardMessage routes around peers whose
// fresh state provably excludes a topic.
type peerState struct {
topics *util.BloomFilter
updatedAt time.Time
}
// peerQueue is the bounded, batching send queue for a single peer, pinned to the advertise URL
// the peer was created with: a peer re-registering under a different advertise URL is treated
// as a replacement (reconcile retires the old queue; ForwardMessage creates a fresh one on demand).
type peerQueue struct {
advertiseURL string
queue *util.LingerQueue[[]byte] // pre-marshaled apiMessage fragments
}
+69
View File
@@ -0,0 +1,69 @@
package cluster
import (
"bufio"
"bytes"
"encoding/json"
"io"
"net/netip"
"strings"
"heckel.io/ntfy/v2/log"
"heckel.io/ntfy/v2/model"
)
// messageURL derives the peer's message endpoint URL from its advertise URL.
func messageURL(advertiseURL string) string {
return strings.TrimRight(advertiseURL, "/") + MessagePath
}
// stateURL derives the peer's state endpoint URL from its advertise URL.
func stateURL(advertiseURL string) string {
return strings.TrimRight(advertiseURL, "/") + StatePath
}
// marshalMessage serializes one message and its non-JSON fields (Sender, User) as an
// apiMessage line. Lines are marshaled once per publish and shared across all per-peer
// queues; assembleMessageBody joins them without re-marshaling.
func marshalMessage(m *model.Message) ([]byte, error) {
apiMsg := &apiMessage{User: m.User, Message: m}
if m.Sender.IsValid() {
apiMsg.Sender = m.Sender.String()
}
return json.Marshal(apiMsg)
}
// assembleMessageBody builds an NDJSON fan-out request body from pre-marshaled apiMessage
// lines, avoiding a second JSON marshal of the messages.
func assembleMessageBody(frags [][]byte) []byte {
return append(bytes.Join(frags, []byte("\n")), '\n')
}
// decodeMessageBody reads NDJSON apiMessage lines from r, reattaches the non-JSON fields
// (Sender, User) onto each message, and hands them to deliver. Malformed or message-less lines
// are skipped and logged, not fatal: fan-out is fire-and-forget, so the valid remainder of a
// request is still delivered. It returns an error only for stream-level failures (e.g. a line
// exceeding maxLineBytes).
func decodeMessageBody(r io.Reader, maxLineBytes int, deliver DeliverFunc) error {
scanner := bufio.NewScanner(r)
scanner.Buffer(make([]byte, 64*1024), maxLineBytes)
for scanner.Scan() {
line := bytes.TrimSpace(scanner.Bytes())
if len(line) == 0 {
continue
}
var apiMsg apiMessage
if err := json.Unmarshal(line, &apiMsg); err != nil || apiMsg.Message == nil {
log.Tag(tag).Warn("Skipping malformed fan-out line")
continue
}
apiMsg.Message.User = apiMsg.User
if apiMsg.Sender != "" {
if addr, err := netip.ParseAddr(apiMsg.Sender); err == nil {
apiMsg.Message.Sender = addr
}
}
deliver(apiMsg.Message)
}
return scanner.Err()
}
+41 -1
View File
@@ -19,6 +19,7 @@ import (
"github.com/urfave/cli/v2"
"github.com/urfave/cli/v2/altsrc"
"heckel.io/ntfy/v2/ban"
"heckel.io/ntfy/v2/cluster"
"heckel.io/ntfy/v2/log"
"heckel.io/ntfy/v2/payments"
"heckel.io/ntfy/v2/server"
@@ -43,6 +44,11 @@ var flagsServe = append(
altsrc.NewStringFlag(&cli.StringFlag{Name: "firebase-key-file", Aliases: []string{"firebase_key_file", "F"}, EnvVars: []string{"NTFY_FIREBASE_KEY_FILE"}, Usage: "Firebase credentials file; if set additionally publish to FCM topic"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "database-url", Aliases: []string{"database_url"}, EnvVars: []string{"NTFY_DATABASE_URL"}, Usage: "PostgreSQL connection string for database-backed stores (e.g. postgres://user:pass@host:5432/ntfy)"}),
altsrc.NewStringSliceFlag(&cli.StringSliceFlag{Name: "database-replica-urls", Aliases: []string{"database_replica_urls"}, EnvVars: []string{"NTFY_DATABASE_REPLICA_URLS"}, Usage: "PostgreSQL read replica connection strings for offloading read queries"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "cluster-node-id", Aliases: []string{"cluster_node_id"}, EnvVars: []string{"NTFY_CLUSTER_NODE_ID"}, Usage: "stable per-node identifier for the cluster node registry (required in cluster mode)"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "cluster-listen", Aliases: []string{"cluster_listen"}, EnvVars: []string{"NTFY_CLUSTER_LISTEN"}, Usage: "ip:port for the dedicated cluster fan-out listener; bind it to the private network (e.g. 10.0.0.5:2587)"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "cluster-advertise-url", Aliases: []string{"cluster_advertise_url"}, EnvVars: []string{"NTFY_CLUSTER_ADVERTISE_URL"}, Usage: "base URL peer nodes use to reach this node's fan-out listener (defaults to http://<cluster-listen>)"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "cluster-secret", Aliases: []string{"cluster_secret"}, EnvVars: []string{"NTFY_CLUSTER_SECRET"}, Usage: "shared secret authenticating node-to-node fan-out requests"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "cluster-batch-linger", Aliases: []string{"cluster_batch_linger"}, EnvVars: []string{"NTFY_CLUSTER_BATCH_LINGER"}, Value: util.FormatDuration(cluster.DefaultBatchLinger), Usage: "how long fan-out messages wait to form a batch per peer node (0 = send immediately)"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "cache-file", Aliases: []string{"cache_file", "C"}, EnvVars: []string{"NTFY_CACHE_FILE"}, Usage: "cache file used for message caching"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "cache-duration", Aliases: []string{"cache_duration", "b"}, EnvVars: []string{"NTFY_CACHE_DURATION"}, Value: util.FormatDuration(server.DefaultCacheDuration), Usage: "buffer messages for this time to allow `since` requests"}),
altsrc.NewIntFlag(&cli.IntFlag{Name: "cache-batch-size", Aliases: []string{"cache_batch_size"}, EnvVars: []string{"NTFY_BATCH_SIZE"}, Usage: "max size of messages to batch together when writing to message cache (if zero, writes are synchronous)"}),
@@ -89,7 +95,7 @@ var flagsServe = append(
altsrc.NewIntFlag(&cli.IntFlag{Name: "visitor-subscription-limit", Aliases: []string{"visitor_subscription_limit"}, EnvVars: []string{"NTFY_VISITOR_SUBSCRIPTION_LIMIT"}, Value: server.DefaultVisitorSubscriptionLimit, Usage: "number of subscriptions per visitor"}),
altsrc.NewBoolFlag(&cli.BoolFlag{Name: "visitor-subscriber-rate-limiting", Aliases: []string{"visitor_subscriber_rate_limiting"}, EnvVars: []string{"NTFY_VISITOR_SUBSCRIBER_RATE_LIMITING"}, Value: false, Usage: "enables subscriber-based rate limiting"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "visitor-attachment-total-size-limit", Aliases: []string{"visitor_attachment_total_size_limit"}, EnvVars: []string{"NTFY_VISITOR_ATTACHMENT_TOTAL_SIZE_LIMIT"}, Value: util.FormatSize(server.DefaultVisitorAttachmentTotalSizeLimit), Usage: "total storage limit used for attachments per visitor"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "visitor-attachment-daily-bandwidth-limit", Aliases: []string{"visitor_attachment_daily_bandwidth_limit"}, EnvVars: []string{"NTFY_VISITOR_ATTACHMENT_DAILY_BANDWIDTH_LIMIT"}, Value: "500M", Usage: "total daily bandwidth limit per visitor, for attachment downloads/uploads and messages replayed from the cache by poll requests"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "visitor-attachment-daily-bandwidth-limit", Aliases: []string{"visitor_attachment_daily_bandwidth_limit"}, EnvVars: []string{"NTFY_VISITOR_ATTACHMENT_DAILY_BANDWIDTH_LIMIT"}, Value: "500M", Usage: "total daily attachment download/upload bandwidth limit per visitor"}),
altsrc.NewIntFlag(&cli.IntFlag{Name: "visitor-request-limit-burst", Aliases: []string{"visitor_request_limit_burst"}, EnvVars: []string{"NTFY_VISITOR_REQUEST_LIMIT_BURST"}, Value: server.DefaultVisitorRequestLimitBurst, Usage: "initial limit of requests per visitor"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "visitor-request-limit-replenish", Aliases: []string{"visitor_request_limit_replenish"}, EnvVars: []string{"NTFY_VISITOR_REQUEST_LIMIT_REPLENISH"}, Value: util.FormatDuration(server.DefaultVisitorRequestLimitReplenish), Usage: "interval at which burst limit is replenished (one per x)"}),
altsrc.NewStringFlag(&cli.StringFlag{Name: "visitor-request-limit-exempt-hosts", Aliases: []string{"visitor_request_limit_exempt_hosts"}, EnvVars: []string{"NTFY_VISITOR_REQUEST_LIMIT_EXEMPT_HOSTS"}, Value: "", Usage: "hostnames and/or IP addresses of hosts that will be exempt from the visitor request limit"}),
@@ -157,6 +163,11 @@ func execServe(c *cli.Context) error {
firebaseKeyFile := c.String("firebase-key-file")
databaseURL := c.String("database-url")
databaseReplicaURLs := c.StringSlice("database-replica-urls")
clusterNodeID := c.String("cluster-node-id")
clusterListen := c.String("cluster-listen")
clusterAdvertiseURL := c.String("cluster-advertise-url")
clusterSecret := c.String("cluster-secret")
clusterBatchLingerStr := c.String("cluster-batch-linger")
webPushPrivateKey := c.String("web-push-private-key")
webPushPublicKey := c.String("web-push-public-key")
webPushFile := c.String("web-push-file")
@@ -252,6 +263,10 @@ func execServe(c *cli.Context) error {
if err != nil {
return fmt.Errorf("invalid keepalive interval: %s", keepaliveIntervalStr)
}
clusterBatchLinger, err := util.ParseDuration(clusterBatchLingerStr)
if err != nil || clusterBatchLinger < 0 {
return fmt.Errorf("invalid cluster batch linger: %s", clusterBatchLingerStr)
}
managerInterval, err := util.ParseDuration(managerIntervalStr)
if err != nil {
return fmt.Errorf("invalid manager interval: %s", managerIntervalStr)
@@ -322,6 +337,16 @@ func execServe(c *cli.Context) error {
return errors.New("if database-url is set, auth-file, cache-file, and web-push-file must not be set")
} else if len(databaseReplicaURLs) > 0 && databaseURL == "" {
return errors.New("database-replica-urls can only be used if database-url is also set")
} else if clusterListen != "" && databaseURL == "" {
return errors.New("cluster-listen requires database-url to be set")
} else if clusterListen != "" && clusterSecret == "" {
return errors.New("cluster-listen requires cluster-secret to be set")
} else if clusterListen != "" && clusterNodeID == "" {
return errors.New("cluster-listen requires cluster-node-id to be set")
} else if clusterListen == "" && clusterSecret != "" {
return errors.New("cluster-secret can only be used if cluster-listen is set")
} else if clusterListen != "" && clusterAdvertiseURL == "" && wildcardAddr(clusterListen) {
return errors.New("cluster-advertise-url must be set if cluster-listen binds a wildcard address")
} else if firebaseKeyFile != "" && !util.FileExists(firebaseKeyFile) {
return errors.New("if set, FCM key file must exist")
} else if firebaseKeyFile != "" && !server.FirebaseAvailable {
@@ -559,6 +584,11 @@ func execServe(c *cli.Context) error {
conf.ProfileListenHTTP = profileListenHTTP
conf.DatabaseURL = databaseURL
conf.DatabaseReplicaURLs = databaseReplicaURLs
conf.ClusterNodeID = clusterNodeID
conf.ClusterListen = clusterListen
conf.ClusterAdvertiseURL = clusterAdvertiseURL
conf.ClusterSecret = clusterSecret
conf.ClusterBatchLinger = clusterBatchLinger
conf.WebPushPrivateKey = webPushPrivateKey
conf.WebPushPublicKey = webPushPublicKey
conf.WebPushFile = webPushFile
@@ -737,3 +767,13 @@ func maybeFromMetadata(m map[string]any, key string) string {
}
return s
}
// wildcardAddr reports whether the given listen address binds all interfaces (e.g. ":2587",
// "0.0.0.0:2587", "[::]:2587"), in which case peers cannot derive a reachable URL from it.
func wildcardAddr(addr string) bool {
host, _, err := net.SplitHostPort(addr)
if err != nil {
return true // Unparseable -> cannot derive a URL either
}
return host == "" || host == "0.0.0.0" || host == "::"
}
+35
View File
@@ -536,6 +536,41 @@ func TestIP_Host_Parsing(t *testing.T) {
}
}
func TestCLI_Serve_ClusterValidation(t *testing.T) {
configFile := newEmptyFile(t) // Avoid issues with existing server.yml file on system
// Setting cluster-listen implicitly enables clustering, which requires database-url; all
// validation must fail before any database connection is attempted
app, _, _, _ := newTestApp()
err := app.Run([]string{"ntfy", "serve", "--config=" + configFile, "--cluster-listen=127.0.0.1:2587"})
require.Error(t, err)
require.Contains(t, err.Error(), "database-url")
// cluster-listen requires cluster-secret
app, _, _, _ = newTestApp()
err = app.Run([]string{"ntfy", "serve", "--config=" + configFile, "--cluster-listen=127.0.0.1:2587", "--database-url=postgres://user:pass@localhost:1/na"})
require.Error(t, err)
require.Contains(t, err.Error(), "cluster-secret")
// cluster-listen requires an explicit stable node ID
app, _, _, _ = newTestApp()
err = app.Run([]string{"ntfy", "serve", "--config=" + configFile, "--cluster-listen=127.0.0.1:2587", "--database-url=postgres://user:pass@localhost:1/na", "--cluster-secret=s3cret"})
require.Error(t, err)
require.Contains(t, err.Error(), "cluster-node-id")
// cluster-secret without cluster-listen is a config error (clustering would silently be off)
app, _, _, _ = newTestApp()
err = app.Run([]string{"ntfy", "serve", "--config=" + configFile, "--cluster-secret=s3cret"})
require.Error(t, err)
require.Contains(t, err.Error(), "cluster-listen")
// A wildcard cluster-listen bind cannot derive an advertise URL
app, _, _, _ = newTestApp()
err = app.Run([]string{"ntfy", "serve", "--config=" + configFile, "--cluster-listen=:2587", "--database-url=postgres://user:pass@localhost:1/na", "--cluster-secret=s3cret", "--cluster-node-id=node-a"})
require.Error(t, err)
require.Contains(t, err.Error(), "cluster-advertise-url")
// cluster-batch-linger must not be negative
app, _, _, _ = newTestApp()
err = app.Run([]string{"ntfy", "serve", "--config=" + configFile, "--cluster-batch-linger=-1s"})
require.Error(t, err)
require.Contains(t, err.Error(), "cluster batch linger")
}
func newEmptyFile(t *testing.T) string {
filename := filepath.Join(t.TempDir(), "empty")
require.Nil(t, os.WriteFile(filename, []byte{}, 0600))
+163
View File
@@ -0,0 +1,163 @@
package pg
import (
"context"
"database/sql"
"sync"
"time"
"heckel.io/ntfy/v2/log"
)
const (
tagLeader = "leader"
tryAdvisoryLockQuery = `SELECT pg_try_advisory_lock($1)`
advisoryUnlockQuery = `SELECT pg_advisory_unlock($1)`
defaultRenewInterval = 5 * time.Second
leaderMissedRenewals = 3
leaderHoldoffFactor = 2
)
// Leader implements singleton-job leader election via a Postgres advisory lock held on a
// pinned connection. The lock auto-releases when the holding connection dies, so a crashed
// leader is replaced without manual fencing; distinct keys elect independently. The Leader
// renews its lease on its own loop; callers only ask IsLeader and eventually Close.
//
// Holding the lock is not the same as believing to be the leader: IsLeader also requires a
// recent renewal (lease duration) and a completed hold-off after winning the lock. The
// hold-off outlasts the lease duration by construction, so on failover the old belief always
// expires before the new one begins: a short no-leader gap, never two leaders. Defaults:
// renew every 5s, lease duration 15s, hold-off 30s -> up to ~35s without a leader.
type Leader struct {
db *sql.DB
key int64
renewInterval time.Duration
conn *sql.Conn // holds the advisory lock while this process is leader
acquiredAt time.Time // When the lock was won (this tenure), for the hold-off
renewedAt time.Time // Last successful renewal, for the lease duration; zero = lock not held
cancel context.CancelFunc // Stops the renew loop and aborts its in-flight query on Close
closeOnce sync.Once
wg sync.WaitGroup
mu sync.Mutex // Protects conn, acquiredAt and renewedAt
}
// NewLeader creates a Leader competing for the lock identified by key and starts its renew
// loop. renewInterval is for tests; pass 0 for the default.
func NewLeader(db *sql.DB, key int64, renewInterval time.Duration) *Leader {
if renewInterval <= 0 {
renewInterval = defaultRenewInterval
}
ctx, cancel := context.WithCancel(context.Background())
l := &Leader{
db: db,
key: key,
renewInterval: renewInterval,
cancel: cancel,
}
l.wg.Add(1)
go l.runAcquireOrRenewLoop(ctx)
return l
}
// IsLeader reports whether this process should act as the leader: lock held, lease renewed
// recently, hold-off elapsed (see the Leader doc comment).
func (l *Leader) IsLeader() bool {
l.mu.Lock()
defer l.mu.Unlock()
leaseDuration := leaderMissedRenewals * l.renewInterval
holdoff := leaderHoldoffFactor * leaseDuration
return time.Since(l.renewedAt) < leaseDuration && time.Since(l.acquiredAt) >= holdoff
}
// Close stops competing for leadership and releases the lock. Idempotent.
func (l *Leader) Close() {
l.closeOnce.Do(func() {
l.cancel() // Also aborts an in-flight renewal query
l.wg.Wait()
if l.IsLeader() {
log.Tag(tagLeader).Info("Lost leadership: closed (lock key %d)", l.key)
}
l.release()
})
}
// runAcquireOrRenewLoop acquires or renews the lock every renewInterval until ctx is canceled
func (l *Leader) runAcquireOrRenewLoop(ctx context.Context) {
defer l.wg.Done()
ticker := time.NewTicker(l.renewInterval)
defer ticker.Stop()
wasLeader := false
for {
attemptCtx, cancel := context.WithTimeout(ctx, l.renewInterval)
l.tryAcquireOrRenew(attemptCtx)
cancel()
if isLeader := l.IsLeader(); isLeader != wasLeader {
wasLeader = isLeader
if isLeader {
log.Tag(tagLeader).Info("Became leader (lock key %d)", l.key)
} else {
log.Tag(tagLeader).Info("Lost leadership (lock key %d)", l.key)
}
}
select {
case <-ticker.C:
case <-ctx.Done():
return
}
}
}
// tryAcquireOrRenew renews the lock on a healthy leader (a cheap ping) or retries acquiring
// it on a follower, on a pinned connection.
func (l *Leader) tryAcquireOrRenew(ctx context.Context) {
l.mu.Lock()
conn := l.conn
l.mu.Unlock()
if conn != nil {
if conn.PingContext(ctx) == nil {
// Still holding the lock, connection healthy: renew the lease
l.mu.Lock()
l.renewedAt = time.Now()
l.mu.Unlock()
log.Tag(tagLeader).Trace("Renewed leader lease (lock key %d)", l.key)
return
}
log.Tag(tagLeader).Debug("Leader lock connection died, lock lost (lock key %d)", l.key)
l.release() // Connection died; the lock is already gone, re-acquire below
}
newConn, err := l.db.Conn(ctx)
if err != nil {
log.Tag(tagLeader).Debug("Cannot get connection to compete for leader lock (lock key %d): %s", l.key, err.Error())
return
}
var acquired bool
if err := newConn.QueryRowContext(ctx, tryAdvisoryLockQuery, l.key).Scan(&acquired); err != nil || !acquired {
newConn.Close()
log.Tag(tagLeader).Trace("Leader lock held elsewhere (lock key %d)", l.key)
return
}
log.Tag(tagLeader).Debug("Acquired leader lock (lock key %d); leadership after the hold-off", l.key)
l.mu.Lock()
l.conn = newConn
l.acquiredAt = time.Now()
l.renewedAt = l.acquiredAt
l.mu.Unlock()
}
// release unlocks the advisory lock and returns the pinned connection to the pool
func (l *Leader) release() {
l.mu.Lock()
conn := l.conn
l.conn = nil
l.renewedAt = time.Time{} // Zero revokes belief; without it, IsLeader would linger a lease duration
l.mu.Unlock()
if conn != nil {
// Unlock explicitly: sql.Conn.Close() returns the connection to the pool, so the
// session-scoped lock would otherwise stay held
conn.ExecContext(context.Background(), advisoryUnlockQuery, l.key)
conn.Close()
log.Tag(tagLeader).Debug("Released leader lock (lock key %d)", l.key)
}
}
+41
View File
@@ -0,0 +1,41 @@
package pg
import (
"testing"
"time"
"github.com/stretchr/testify/require"
)
// The lease logic is pure time arithmetic, so it is unit-tested here without a database; the
// external leader tests cover the loop end to end.
func TestLeader_Lease_HoldoffMeansNoLeaderRatherThanTwo(t *testing.T) {
// Freshly acquired lock: belief must wait out the hold-off
l := &Leader{renewInterval: 20 * time.Second} // Lease duration 1m, hold-off 2m
l.acquiredAt = time.Now()
l.renewedAt = l.acquiredAt
require.False(t, l.IsLeader())
// Once the hold-off has passed (and verification is fresh), belief begins
l.acquiredAt = time.Now().Add(-3 * time.Minute)
l.renewedAt = time.Now()
require.True(t, l.IsLeader())
}
func TestLeader_Lease_ExpiredLeaseRevokesLeadership(t *testing.T) {
// A leader that cannot renew its lease (wedged process, long GC pause) must stop
// believing once the lease expires, even though the lock may still be held
l := &Leader{renewInterval: 20 * time.Second} // Lease duration 1m, hold-off 2m
l.acquiredAt = time.Now().Add(-time.Hour)
l.renewedAt = time.Now().Add(-2 * time.Minute) // Lease expired
require.False(t, l.IsLeader())
l.renewedAt = time.Now() // Fresh renewal restores belief
require.True(t, l.IsLeader())
}
func TestLeader_Lease_ReleasedIsNeverLeader(t *testing.T) {
// release() zeroes renewedAt, which fails the lease check no matter how old the tenure
l := &Leader{renewInterval: 20 * time.Second} // Lease duration 1m, hold-off 2m
l.acquiredAt = time.Now().Add(-time.Hour)
require.False(t, l.IsLeader())
}
+90
View File
@@ -0,0 +1,90 @@
package pg_test
import (
"testing"
"time"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/db/pg"
dbtest "heckel.io/ntfy/v2/db/test"
)
const testRenewInterval = 20 * time.Millisecond // Lease duration 60ms, hold-off 120ms
func TestLeader_AcquireAndFailover(t *testing.T) {
testDB := dbtest.CreateTestPostgres(t) // skips if NTFY_TEST_DATABASE_URL is unset
const key = int64(42)
l1 := pg.NewLeader(testDB.Primary(), key, testRenewInterval)
defer l1.Close()
// Belief follows the hold-off, it is never instant
require.False(t, l1.IsLeader())
waitForLeader(t, l1)
// A competitor never becomes leader while the leader lives
l2 := pg.NewLeader(testDB.Primary(), key, testRenewInterval)
defer l2.Close()
time.Sleep(300 * time.Millisecond) // Several verification rounds
require.False(t, l2.IsLeader())
require.True(t, l1.IsLeader())
// Close -> the follower takes over
l1.Close()
require.False(t, l1.IsLeader())
waitForLeader(t, l2)
require.False(t, l1.IsLeader())
}
func TestLeader_ConnectionLossFailover(t *testing.T) {
// A crashed leader must not wedge the cluster: Postgres releases the session-scoped lock
// when the pinned connection dies (simulated by terminating the backend), and someone
// re-acquires. Either node may win; the invariant is one leader eventually, never two.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
hostA, err := pg.Open(schemaDSN)
require.Nil(t, err)
defer hostA.DB.Close()
hostB, err := pg.Open(schemaDSN)
require.Nil(t, err)
defer hostB.DB.Close()
const key = int64(43)
l1 := pg.NewLeader(hostA.DB, key, testRenewInterval)
defer l1.Close()
waitForLeader(t, l1)
l2 := pg.NewLeader(hostB.DB, key, testRenewInterval)
defer l2.Close()
// Kill the backend holding the lock (advisory lock keys map to classid/objid)
_, err = hostB.DB.Exec(`SELECT pg_terminate_backend(pid) FROM pg_locks WHERE locktype = 'advisory' AND objid = $1 AND granted`, key)
require.Nil(t, err)
// Eventually exactly one leader again, and never two along the way
deadline := time.Now().Add(5 * time.Second)
for time.Now().Before(deadline) {
leader1, leader2 := l1.IsLeader(), l2.IsLeader()
require.False(t, leader1 && leader2, "two leaders at once")
if leader1 != leader2 {
return
}
time.Sleep(10 * time.Millisecond)
}
t.Fatal("no leader re-emerged after connection loss")
}
func TestLeader_DistinctKeysAreIndependent(t *testing.T) {
testDB := dbtest.CreateTestPostgres(t)
l1 := pg.NewLeader(testDB.Primary(), 1, testRenewInterval)
defer l1.Close()
l2 := pg.NewLeader(testDB.Primary(), 2, testRenewInterval)
defer l2.Close()
// Different keys do not compete: both become effective leaders
waitForLeader(t, l1)
waitForLeader(t, l2)
}
// waitForLeader waits until the node believes it is the leader, or fails the test
func waitForLeader(t *testing.T, l *pg.Leader) {
t.Helper()
deadline := time.Now().Add(5 * time.Second)
for time.Now().Before(deadline) {
if l.IsLeader() {
return
}
time.Sleep(10 * time.Millisecond)
}
t.Fatal("node never became effective leader")
}
+1
View File
@@ -17,6 +17,7 @@ import (
// ntfy key is defined here, following the "ntfy"+2586+letter scheme
const (
SchemaLockKey = int64(0x6e7466792586a) // Schema setup serialization (transaction-scoped, see db/schema)
LeaderLockKey = int64(0x6e7466792586b) // Cluster singleton-job leader (session-scoped, held for process lifetime)
)
// Open opens a PostgreSQL connection pool for a primary database. It pings the database
+3 -1
View File
@@ -14,7 +14,9 @@ import (
_ "github.com/mattn/go-sqlite3"
)
const testCreateQuery = `CREATE TABLE IF NOT EXISTS things (id TEXT PRIMARY KEY, name TEXT NOT NULL)`
const (
testCreateQuery = `CREATE TABLE IF NOT EXISTS things (id TEXT PRIMARY KEY, name TEXT NOT NULL)`
)
func testCreate(tx *sql.Tx) error {
_, err := tx.Exec(testCreateQuery)
+2 -2
View File
@@ -17,7 +17,7 @@ const testPoolMaxConns = "2"
// CreateTestPostgresSchema creates a temporary PostgreSQL schema and returns the DSN pointing to it.
// It registers a cleanup function to drop the schema when the test finishes.
// If NTFY_TEST_DATABASE_URL is not set, the test is skipped.
func CreateTestPostgresSchema(t *testing.T) string {
func CreateTestPostgresSchema(t testing.TB) string {
t.Helper()
dsn := os.Getenv("NTFY_TEST_DATABASE_URL")
if dsn == "" {
@@ -51,7 +51,7 @@ func CreateTestPostgresSchema(t *testing.T) string {
// CreateTestPostgres creates a temporary PostgreSQL schema and returns an open *db.DB connection to it.
// It registers cleanup functions to close the DB and drop the schema when the test finishes.
// If NTFY_TEST_DATABASE_URL is not set, the test is skipped.
func CreateTestPostgres(t *testing.T) *db.DB {
func CreateTestPostgres(t testing.TB) *db.DB {
t.Helper()
schemaDSN := CreateTestPostgresSchema(t)
testHost, err := pg.Open(schemaDSN)
+3 -5
View File
@@ -1911,9 +1911,7 @@ per-visitor limits:
* `visitor-attachment-total-size-limit` is the total storage limit used for attachments per visitor. It defaults to 100M.
The per-visitor storage is automatically decreased as attachments expire. External attachments (attached via `X-Attach`,
see [publishing docs](publish.md#attachments)) do not count here.
* `visitor-attachment-daily-bandwidth-limit` is the total daily bandwidth limit per visitor. It covers attachment
downloads/uploads, and messages replayed from the message cache by poll requests (a poll without a `since` cursor
returns a topic's entire cache, so a busy topic can be re-read for many times its own size),
* `visitor-attachment-daily-bandwidth-limit` is the total daily attachment download/upload bandwidth limit per visitor,
including PUT and GET requests. This is to protect your precious bandwidth from abuse, since egress costs money in
most cloud providers. This defaults to 500M.
@@ -2384,7 +2382,7 @@ variable before running the `ntfy` command (e.g. `export NTFY_LISTEN_HTTP=:80`).
| `upstream-base-url` | `NTFY_UPSTREAM_BASE_URL` | *URL* | `https://ntfy.sh` | Forward poll request to an upstream server, this is needed for iOS push notifications for self-hosted servers |
| `upstream-access-token` | `NTFY_UPSTREAM_ACCESS_TOKEN` | *string* | `tk_zyYLYj...` | Access token to use for the upstream server; needed only if upstream rate limits are exceeded or upstream server requires auth |
| `visitor-attachment-total-size-limit` | `NTFY_VISITOR_ATTACHMENT_TOTAL_SIZE_LIMIT` | *size* | 100M | Rate limiting: Total storage limit used for attachments per visitor, for all attachments combined. Storage is freed after attachments expire. See `attachment-expiry-duration`. |
| `visitor-attachment-daily-bandwidth-limit` | `NTFY_VISITOR_ATTACHMENT_DAILY_BANDWIDTH_LIMIT` | *size* | 500M | Rate limiting: Total daily traffic limit per visitor, covering attachment downloads/uploads and messages replayed from the cache by poll requests. This is to protect your bandwidth costs from exploding. |
| `visitor-attachment-daily-bandwidth-limit` | `NTFY_VISITOR_ATTACHMENT_DAILY_BANDWIDTH_LIMIT` | *size* | 500M | Rate limiting: Total daily attachment download/upload traffic limit per visitor. This is to protect your bandwidth costs from exploding. |
| `visitor-email-limit-burst` | `NTFY_VISITOR_EMAIL_LIMIT_BURST` | *number* | 16 | Rate limiting:Initial limit of e-mails per visitor |
| `visitor-email-limit-replenish` | `NTFY_VISITOR_EMAIL_LIMIT_REPLENISH` | *duration* | 1h | Rate limiting: Strongly related to `visitor-email-limit-burst`: The rate at which the bucket is refilled |
| `visitor-message-daily-limit` | `NTFY_VISITOR_MESSAGE_DAILY_LIMIT` | *number* | - | Rate limiting: Allowed number of messages per day per visitor, reset every day at midnight (UTC). By default, this value is unset. |
@@ -2500,7 +2498,7 @@ OPTIONS:
--visitor-subscription-limit value, --visitor_subscription_limit value number of subscriptions per visitor (default: 30) [$NTFY_VISITOR_SUBSCRIPTION_LIMIT]
--visitor-subscriber-rate-limiting, --visitor_subscriber_rate_limiting enables subscriber-based rate limiting (default: false) [$NTFY_VISITOR_SUBSCRIBER_RATE_LIMITING]
--visitor-attachment-total-size-limit value, --visitor_attachment_total_size_limit value total storage limit used for attachments per visitor (default: "100M") [$NTFY_VISITOR_ATTACHMENT_TOTAL_SIZE_LIMIT]
--visitor-attachment-daily-bandwidth-limit value, --visitor_attachment_daily_bandwidth_limit value total daily bandwidth limit per visitor, for attachment downloads/uploads and messages replayed from the cache by poll requests (default: "500M") [$NTFY_VISITOR_ATTACHMENT_DAILY_BANDWIDTH_LIMIT]
--visitor-attachment-daily-bandwidth-limit value, --visitor_attachment_daily_bandwidth_limit value total daily attachment download/upload bandwidth limit per visitor (default: "500M") [$NTFY_VISITOR_ATTACHMENT_DAILY_BANDWIDTH_LIMIT]
--visitor-request-limit-burst value, --visitor_request_limit_burst value initial limit of requests per visitor (default: 60) [$NTFY_VISITOR_REQUEST_LIMIT_BURST]
--visitor-request-limit-replenish value, --visitor_request_limit_replenish value interval at which burst limit is replenished (one per x) (default: "5s") [$NTFY_VISITOR_REQUEST_LIMIT_REPLENISH]
--visitor-request-limit-exempt-hosts value, --visitor_request_limit_exempt_hosts value hostnames and/or IP addresses of hosts that will be exempt from the visitor request limit [$NTFY_VISITOR_REQUEST_LIMIT_EXEMPT_HOSTS]
+38 -38
View File
@@ -34,37 +34,37 @@ as a service starting at boot time.
=== "x86_64/amd64"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_amd64.tar.gz
tar zxvf ntfy_2.28.0_linux_amd64.tar.gz
sudo cp -a ntfy_2.28.0_linux_amd64/ntfy /usr/local/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.28.0_linux_amd64/{client,server}/*.yml /etc/ntfy
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_amd64.tar.gz
tar zxvf ntfy_2.27.0_linux_amd64.tar.gz
sudo cp -a ntfy_2.27.0_linux_amd64/ntfy /usr/local/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.27.0_linux_amd64/{client,server}/*.yml /etc/ntfy
sudo ntfy serve
```
=== "armv6"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_armv6.tar.gz
tar zxvf ntfy_2.28.0_linux_armv6.tar.gz
sudo cp -a ntfy_2.28.0_linux_armv6/ntfy /usr/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.28.0_linux_armv6/{client,server}/*.yml /etc/ntfy
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_armv6.tar.gz
tar zxvf ntfy_2.27.0_linux_armv6.tar.gz
sudo cp -a ntfy_2.27.0_linux_armv6/ntfy /usr/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.27.0_linux_armv6/{client,server}/*.yml /etc/ntfy
sudo ntfy serve
```
=== "armv7/armhf"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_armv7.tar.gz
tar zxvf ntfy_2.28.0_linux_armv7.tar.gz
sudo cp -a ntfy_2.28.0_linux_armv7/ntfy /usr/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.28.0_linux_armv7/{client,server}/*.yml /etc/ntfy
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_armv7.tar.gz
tar zxvf ntfy_2.27.0_linux_armv7.tar.gz
sudo cp -a ntfy_2.27.0_linux_armv7/ntfy /usr/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.27.0_linux_armv7/{client,server}/*.yml /etc/ntfy
sudo ntfy serve
```
=== "arm64"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_arm64.tar.gz
tar zxvf ntfy_2.28.0_linux_arm64.tar.gz
sudo cp -a ntfy_2.28.0_linux_arm64/ntfy /usr/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.28.0_linux_arm64/{client,server}/*.yml /etc/ntfy
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_arm64.tar.gz
tar zxvf ntfy_2.27.0_linux_arm64.tar.gz
sudo cp -a ntfy_2.27.0_linux_arm64/ntfy /usr/bin/ntfy
sudo mkdir /etc/ntfy && sudo cp ntfy_2.27.0_linux_arm64/{client,server}/*.yml /etc/ntfy
sudo ntfy serve
```
@@ -84,25 +84,25 @@ Install the ntfy server unit file (which contains parameters to start the servic
=== "x86_64/amd64"
```bash
sudo mv ntfy_2.28.0_linux_amd64/server/ntfy.service /etc/systemd/system/
sudo mv ntfy_2.27.0_linux_amd64/server/ntfy.service /etc/systemd/system/
sudo chmod 644 /etc/systemd/system/ntfy.service
```
=== "armv6"
```bash
sudo mv ntfy_2.28.0_linux_armv6/server/ntfy.service /etc/systemd/system/
sudo mv ntfy_2.27.0_linux_armv6/server/ntfy.service /etc/systemd/system/
sudo chmod 644 /etc/systemd/system/ntfy.service
```
=== "armv7/armhf"
```bash
sudo mv ntfy_2.28.0_linux_armv7/server/ntfy.service /etc/systemd/system/
sudo mv ntfy_2.27.0_linux_armv7/server/ntfy.service /etc/systemd/system/
sudo chmod 644 /etc/systemd/system/ntfy.service
```
=== "arm64"
```bash
sudo mv ntfy_2.28.0_linux_arm64/server/ntfy.service /etc/systemd/system/
sudo mv ntfy_2.27.0_linux_arm64/server/ntfy.service /etc/systemd/system/
sudo chmod 644 /etc/systemd/system/ntfy.service
```
@@ -118,25 +118,25 @@ Install the ntfy server service script:
=== "x86_64/amd64"
```bash
sudo mv ntfy_2.28.0_linux_amd64/server/ntfy.openrc /etc/init.d/ntfy
sudo mv ntfy_2.27.0_linux_amd64/server/ntfy.openrc /etc/init.d/ntfy
sudo chmod 755 /etc/init.d/ntfy
```
=== "armv6"
```bash
sudo mv ntfy_2.28.0_linux_armv6/server/ntfy.openrc /etc/init.d/ntfy
sudo mv ntfy_2.27.0_linux_armv6/server/ntfy.openrc /etc/init.d/ntfy
sudo chmod 755 /etc/init.d/ntfy
```
=== "armv7/armhf"
```bash
sudo mv ntfy_2.28.0_linux_armv7/server/ntfy.openrc /etc/init.d/ntfy
sudo mv ntfy_2.27.0_linux_armv7/server/ntfy.openrc /etc/init.d/ntfy
sudo chmod 755 /etc/init.d/ntfy
```
=== "arm64"
```bash
sudo mv ntfy_2.28.0_linux_arm64/server/ntfy.openrc /etc/init.d/ntfy
sudo mv ntfy_2.27.0_linux_arm64/server/ntfy.openrc /etc/init.d/ntfy
sudo chmod 755 /etc/init.d/ntfy
```
@@ -204,7 +204,7 @@ Manually installing the .deb file:
=== "x86_64/amd64"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_amd64.deb
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_amd64.deb
sudo dpkg -i ntfy_*.deb
sudo systemctl enable ntfy
sudo systemctl start ntfy
@@ -212,7 +212,7 @@ Manually installing the .deb file:
=== "armv6"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_armv6.deb
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_armv6.deb
sudo dpkg -i ntfy_*.deb
sudo systemctl enable ntfy
sudo systemctl start ntfy
@@ -220,7 +220,7 @@ Manually installing the .deb file:
=== "armv7/armhf"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_armv7.deb
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_armv7.deb
sudo dpkg -i ntfy_*.deb
sudo systemctl enable ntfy
sudo systemctl start ntfy
@@ -228,7 +228,7 @@ Manually installing the .deb file:
=== "arm64"
```bash
wget https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_arm64.deb
wget https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_arm64.deb
sudo dpkg -i ntfy_*.deb
sudo systemctl enable ntfy
sudo systemctl start ntfy
@@ -238,28 +238,28 @@ Manually installing the .deb file:
=== "x86_64/amd64"
```bash
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_amd64.rpm
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_amd64.rpm
sudo systemctl enable ntfy
sudo systemctl start ntfy
```
=== "armv6"
```bash
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_armv6.rpm
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_armv6.rpm
sudo systemctl enable ntfy
sudo systemctl start ntfy
```
=== "armv7/armhf"
```bash
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_armv7.rpm
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_armv7.rpm
sudo systemctl enable ntfy
sudo systemctl start ntfy
```
=== "arm64"
```bash
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_linux_arm64.rpm
sudo rpm -ivh https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_linux_arm64.rpm
sudo systemctl enable ntfy
sudo systemctl start ntfy
```
@@ -301,18 +301,18 @@ pkg install go-ntfy
## macOS
The [ntfy CLI](subscribe/cli.md) (`ntfy publish` and `ntfy subscribe` only) is supported on macOS as well.
To install, please [download the tarball](https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_darwin_all.tar.gz),
To install, please [download the tarball](https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_darwin_all.tar.gz),
extract it and place it somewhere in your `PATH` (e.g. `/usr/local/bin/ntfy`).
If run as `root`, ntfy will look for its config at `/etc/ntfy/client.yml`. For all other users, it'll look for it at
`~/Library/Application Support/ntfy/client.yml` (sample included in the tarball).
```bash
curl -L https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_darwin_all.tar.gz > ntfy_2.28.0_darwin_all.tar.gz
tar zxvf ntfy_2.28.0_darwin_all.tar.gz
sudo cp -a ntfy_2.28.0_darwin_all/ntfy /usr/local/bin/ntfy
curl -L https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_darwin_all.tar.gz > ntfy_2.27.0_darwin_all.tar.gz
tar zxvf ntfy_2.27.0_darwin_all.tar.gz
sudo cp -a ntfy_2.27.0_darwin_all/ntfy /usr/local/bin/ntfy
mkdir ~/Library/Application\ Support/ntfy
cp ntfy_2.28.0_darwin_all/client/client.yml ~/Library/Application\ Support/ntfy/client.yml
cp ntfy_2.27.0_darwin_all/client/client.yml ~/Library/Application\ Support/ntfy/client.yml
ntfy --help
```
@@ -333,7 +333,7 @@ brew install ntfy
The ntfy server and CLI are fully supported on Windows. You can run the ntfy server directly or as a Windows service.
To install, you can either
* [Download the latest ZIP](https://github.com/binwiederhier/ntfy/releases/download/v2.28.0/ntfy_2.28.0_windows_amd64.zip),
* [Download the latest ZIP](https://github.com/binwiederhier/ntfy/releases/download/v2.27.0/ntfy_2.27.0_windows_amd64.zip),
extract it and place the `ntfy.exe` binary somewhere in your `%Path%`.
* Or install ntfy from the [Scoop](https://scoop.sh) main repository via `scoop install ntfy`
-1
View File
@@ -85,7 +85,6 @@ I've added a ⭐ to projects or posts that have a significant following, or had
- [ntfy-java](https://github.com/MaheshBabu11/ntfy-java/) - A Java package to interact with a ntfy server (Java)
- [aiontfy](https://github.com/tr4nt0r/aiontfy) - Asynchronous client library for publishing and subscribing to ntfy (Python)
- [ex_ntfy](https://github.com/houllette/ex_ntfy) - Elixir SDK covering publishing, polling, and streaming subscriptions for ntfy servers (Elixir)
- [ntfy-logging](https://github.com/Pimak/ntfy-logging) - Turns JVM error logs into ntfy notifications, with zero-code adapters for java.util.logging, Logback, Log4j2, Spring Boot, Micronaut and Quarkus (Java)
## CLIs + GUIs
+1 -2
View File
@@ -4931,8 +4931,7 @@ but just in case, let's list them all:
| **Subscription limit** | By default, the server allows each visitor to keep 30 connections to the server open. |
| **Attachment size limit** | By default, the server allows attachments up to 15 MB in size, up to 100 MB in total per visitor and up to 5 GB across all visitors. On ntfy.sh, the attachment size limit is 2 MB, and the per-visitor total is 20 MB. |
| **Attachment expiry** | By default, the server deletes attachments after 3 hours and thereby frees up space from the total visitor attachment limit. |
| **Title and tag size** | The message title is limited to 1 KB, and all tags combined to 512 bytes. Requests exceeding either are rejected with HTTP 400. |
| **Daily bandwidth** | By default, the server allows 500 MB of traffic per visitor in a 24 hour period, covering attachment GET/PUT/POST traffic and messages replayed from the cache by [poll requests](subscribe/api.md#replay-limits). Traffic exceeding that is rejected. On ntfy.sh, the daily bandwidth limit is 200 MB. |
| **Attachment bandwidth** | By default, the server allows 500 MB of GET/PUT/POST traffic for attachments per visitor in a 24 hour period. Traffic exceeding that is rejected. On ntfy.sh, the daily bandwidth limit is 200 MB. |
| **Total number of topics** | By default, the server is configured to allow 15,000 topics. The ntfy.sh server has higher limits though. |
These limits can be changed on a per-user basis using [tiers](config.md#tiers). If [payments](config.md#payments) are enabled, a user tier can be changed by purchasing
+1 -16
View File
@@ -6,27 +6,12 @@ and the [ntfy Android app](https://github.com/binwiederhier/ntfy-android/release
| Component | Version | Release date |
|------------------|---------|---------------|
| ntfy server | v2.28.0 | Aug 27, 2026 |
| ntfy server | v2.27.0 | Aug 4, 2026 |
| ntfy Android app | v1.25.2 | July 23, 2026 |
| ntfy iOS app | v1.7.0 | May 30, 2026 |
Please check out the release notes for [upcoming releases](#not-released-yet) below.
### ntfy server v2.28.0
Released August 27, 2026
This is a hardening release. A single topic on ntfy.sh was polled continuously with `poll=1` and no
`since` cursor, which replays a topic's entire cache on every request. The changes below bound what one
replay can cost, close two fields that had no size limit at all, and fix an ordering bug found while
digging into it.
**Bug fixes + maintenance:**
* Fix messages being returned out of publish order when polling or replaying **several topics at once** (`/topic1,topic2/json?poll=1`). `Message.Time` has second granularity, so a multi-topic replay sorts many equal keys; the sort was unstable, which could shuffle a single topic's own messages. Single-topic replays were not affected ([#1297](https://github.com/binwiederhier/ntfy/issues/1297))
* Limit the message title to 1 KB and all tags combined to 512 bytes, rejecting larger requests with HTTP 400 (error codes `40057` and `40058`). Neither field had a size limit before, unlike the message body; on ntfy.sh the 99.9th percentile is 212 bytes for titles and 244 for tags
* Cap a single cache replay at 10 MB of messages per topic. A poll without a `since` cursor returns a topic's entire cache, which was previously unbounded and could reach tens of megabytes on a busy topic, so one request could allocate that much on the server. The newest messages that fit are kept and a truncated response carries an `X-Messages-Truncated: 1` header
* `visitor-attachment-daily-bandwidth-limit` now also covers messages replayed from the message cache by poll requests, not just attachment traffic. A poll without a `since` cursor returns a topic's entire cache, so a topic that is cheap to fill can be re-read for many times its own size; polls beyond the budget are rejected with HTTP 429 (error code 42905) before anything is written. **Note that heavy pollers now consume the same budget as attachment downloads**, so operators serving both may want to raise the limit
### ntfy server v2.27.0
Released August 4, 2026
+1 -31
View File
@@ -245,9 +245,6 @@ combined with `since=` (defaults to `since=all`).
curl -s "ntfy.sh/mytopic/json?poll=1"
```
Note that a poll without `since=` returns a topic's **entire cache**, which on a busy topic can be
large. See [replay limits](#replay-limits) below.
### Fetch cached messages
Messages may be cached for a couple of hours (see [message caching](../config.md#message-cache)) to account for network
interruptions of subscribers. If the server has configured message caching, you can read back what you missed by using
@@ -278,28 +275,6 @@ parameter (makes most sense with the `poll=1` parameter):
curl -s "ntfy.sh/mytopic/json?poll=1&sched=1"
```
### Replay limits
Reading cached messages (a `poll=1` request, or any request with `since=`) replays messages the server
already stored, so unlike a live subscription its cost grows with the size of the topic's cache. Two
server-side limits apply, both of which a well-behaved client should handle:
* **The response is size-capped.** A replay returns only the newest messages that fit in 10 MB
**per topic** (counting body, title, tags and every other publisher-set field), and a capped response carries an `X-Messages-Truncated: 1` header. If you see that
header, older messages were dropped and you did not receive the full cache. In practice this only
affects very large topics; a client polling with `since=` never comes close.
* **Replayed bytes count against your daily bandwidth budget**, the same one attachment downloads use
(see [limitations](../publish.md#limitations)). Exceeding it returns `HTTP 429` with ntfy error code
`42905`, and no messages are written.
Both limits exist because a poll without `since=` re-reads the whole cache every time. If you are
polling repeatedly, **pass `since=<last message ID>`** rather than re-fetching everything. The limits
still apply to a `since=` replay, but it returns only what is new, so in practice you will not come
near either one:
```
curl -s "ntfy.sh/mytopic/json?poll=1&since=nFS3knfcQ1xe"
```
### Filter messages
You can filter which messages are returned based on the well-known message fields `id`, `message`, `title`, `priority` and
`tags`. Here's an example that only returns messages of high or urgent priority that contains the both tags
@@ -333,11 +308,6 @@ $ curl -s ntfy.sh/mytopic1,mytopic2/json
{"id":"Cm02DsxUHb","time":1637182643,"event":"message","topic":"mytopic2","message":"for topic 2"}
```
When replaying cached messages for several topics at once, they are ordered by their `time` field.
Because `time` has **second granularity**, messages published within the same second share a sort key:
each topic's own messages stay in publish order, but the interleaving *between* topics is not defined.
If you need a total order across topics, sort by `time` and fall back to the order received.
### Authentication
Depending on whether the server is configured to support [access control](../config.md#access-control), some topics
may be read/write protected so that only users with the correct credentials can subscribe or publish to them.
@@ -457,7 +427,7 @@ and can be passed as **HTTP headers** or **query parameters in the URL**. They a
| Parameter | Aliases (case-insensitive) | Description |
|-------------|----------------------------|---------------------------------------------------------------------------------|
| `poll` | `X-Poll`, `po` | Return cached messages and close connection (see [replay limits](#replay-limits)) |
| `poll` | `X-Poll`, `po` | Return cached messages and close connection |
| `since` | `X-Since`, `si` | Return cached messages since timestamp, duration or message ID |
| `scheduled` | `X-Scheduled`, `sched` | Include scheduled/delayed messages in message list |
| `id` | `X-ID` | Filter: Only return messages that match this exact message ID |
+36 -36
View File
@@ -1,25 +1,25 @@
module heckel.io/ntfy/v2
go 1.26.0
go 1.25.8
require (
cloud.google.com/go/firestore v1.25.0 // indirect
cloud.google.com/go/storage v1.65.1 // indirect
cloud.google.com/go/firestore v1.24.0 // indirect
cloud.google.com/go/storage v1.64.0 // indirect
github.com/BurntSushi/toml v1.6.0 // indirect
github.com/cpuguy83/go-md2man/v2 v2.0.7 // indirect
github.com/emersion/go-smtp v0.25.0
github.com/emersion/go-smtp v0.24.0
github.com/gabriel-vasile/mimetype v1.4.15
github.com/gorilla/websocket v1.5.3
github.com/mattn/go-sqlite3 v1.14.50
github.com/mattn/go-sqlite3 v1.14.49
github.com/olebedev/when v1.1.0
github.com/stretchr/testify v1.12.1
github.com/stretchr/testify v1.11.1
github.com/urfave/cli/v2 v2.27.7
golang.org/x/crypto v0.55.0
golang.org/x/crypto v0.54.0
golang.org/x/oauth2 v0.36.0 // indirect
golang.org/x/sync v0.22.0
golang.org/x/term v0.45.0
golang.org/x/time v0.15.0
google.golang.org/api v0.294.0
google.golang.org/api v0.291.0
gopkg.in/yaml.v2 v2.4.0
)
@@ -34,30 +34,31 @@ require (
github.com/microcosm-cc/bluemonday v1.0.27
github.com/prometheus/client_golang v1.24.1
github.com/stripe/stripe-go/v74 v74.30.0
golang.org/x/sys v0.48.0
golang.org/x/text v0.41.0
golang.org/x/sys v0.47.0
golang.org/x/text v0.40.0
)
require (
cel.dev/expr v0.25.3 // indirect
cel.dev/expr v0.25.2 // indirect
cloud.google.com/go v0.123.0 // indirect
cloud.google.com/go/auth v0.23.2 // indirect
cloud.google.com/go/auth v0.22.0 // indirect
cloud.google.com/go/auth/oauth2adapt v0.2.8 // indirect
cloud.google.com/go/compute/metadata v0.9.0 // indirect
cloud.google.com/go/iam v1.13.0 // indirect
cloud.google.com/go/iam v1.12.0 // indirect
cloud.google.com/go/longrunning v1.2.0 // indirect
cloud.google.com/go/monitoring v1.30.0 // indirect
github.com/AlekSi/pointer v1.2.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.36.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.60.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.60.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.59.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.59.0 // indirect
github.com/MicahParks/keyfunc v1.9.0 // indirect
github.com/aymerick/douceur v0.2.0 // indirect
github.com/beorn7/perks v1.0.1 // indirect
github.com/cespare/xxhash/v2 v2.3.0 // indirect
github.com/cncf/xds/go v0.0.0-20260202195803-dba9d589def2 // indirect
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc // indirect
github.com/emersion/go-sasl v0.0.0-20241020182733-b788ff22d5a6 // indirect
github.com/envoyproxy/go-control-plane/envoy v1.39.0 // indirect
github.com/envoyproxy/go-control-plane/envoy v1.37.0 // indirect
github.com/envoyproxy/protoc-gen-validate v1.3.3 // indirect
github.com/felixge/httpsnoop v1.1.0 // indirect
github.com/go-jose/go-jose/v4 v4.1.4 // indirect
@@ -68,38 +69,37 @@ require (
github.com/golang/protobuf v1.5.4 // indirect
github.com/google/s2a-go v0.1.9 // indirect
github.com/google/uuid v1.6.0 // indirect
github.com/googleapis/enterprise-certificate-proxy v0.3.21 // indirect
github.com/googleapis/gax-go/v2 v2.24.0 // indirect
github.com/googleapis/enterprise-certificate-proxy v0.3.19 // indirect
github.com/googleapis/gax-go/v2 v2.23.0 // indirect
github.com/gorilla/css v1.0.1 // indirect
github.com/jackc/pgpassfile v1.0.0 // indirect
github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect
github.com/jackc/puddle/v2 v2.2.2 // indirect
github.com/kr/text v0.2.0 // indirect
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect
github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 // indirect
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect
github.com/prometheus/client_model v0.6.2 // indirect
github.com/prometheus/common v0.70.1 // indirect
github.com/prometheus/procfs v0.21.1 // indirect
github.com/russross/blackfriday/v2 v2.1.0 // indirect
github.com/spiffe/go-spiffe/v2 v2.8.1 // indirect
github.com/stretchr/objx v0.5.3 // indirect
github.com/stretchr/objx v0.5.2 // indirect
github.com/xrash/smetrics v0.0.0-20250705151800-55b8f293f342 // indirect
go.opentelemetry.io/auto/sdk v1.2.1 // indirect
go.opentelemetry.io/contrib/detectors/gcp v1.46.0 // indirect
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.71.0 // indirect
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0 // indirect
go.opentelemetry.io/otel v1.46.0 // indirect
go.opentelemetry.io/otel/metric v1.46.0 // indirect
go.opentelemetry.io/otel/sdk v1.46.0 // indirect
go.opentelemetry.io/otel/sdk/metric v1.46.0 // indirect
go.opentelemetry.io/otel/trace v1.46.0 // indirect
go.yaml.in/yaml/v3 v3.0.5 // indirect
golang.org/x/net v0.58.0 // indirect
go.opentelemetry.io/contrib/detectors/gcp v1.44.0 // indirect
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0 // indirect
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 // indirect
go.opentelemetry.io/otel v1.44.0 // indirect
go.opentelemetry.io/otel/metric v1.44.0 // indirect
go.opentelemetry.io/otel/sdk v1.44.0 // indirect
go.opentelemetry.io/otel/sdk/metric v1.44.0 // indirect
go.opentelemetry.io/otel/trace v1.44.0 // indirect
golang.org/x/net v0.57.0 // indirect
google.golang.org/appengine/v2 v2.0.6 // indirect
google.golang.org/genproto v0.0.0-20260825221802-da73d73af1c5 // indirect
google.golang.org/genproto/googleapis/api v0.0.0-20260825221802-da73d73af1c5 // indirect
google.golang.org/genproto/googleapis/rpc v0.0.0-20260825221802-da73d73af1c5 // indirect
google.golang.org/grpc v1.83.2 // indirect
google.golang.org/protobuf v1.36.12 // indirect
google.golang.org/genproto v0.0.0-20260803160001-6ac0973c030d // indirect
google.golang.org/genproto/googleapis/api v0.0.0-20260803160001-6ac0973c030d // indirect
google.golang.org/genproto/googleapis/rpc v0.0.0-20260803160001-6ac0973c030d // indirect
google.golang.org/grpc v1.83.0 // indirect
google.golang.org/protobuf v1.36.11 // indirect
gopkg.in/yaml.v3 v3.0.1 // indirect
)
+74 -73
View File
@@ -1,25 +1,25 @@
cel.dev/expr v0.25.3 h1:A2jO8jwOugrrovveCWfj0KEZOfqiLgAcwjpHPhzIGw0=
cel.dev/expr v0.25.3/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
cel.dev/expr v0.25.2 h1:K6j46C81hXtZQfuX60cVWQFBJahKSE2gfRbNuvr5bFs=
cel.dev/expr v0.25.2/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
cloud.google.com/go v0.123.0 h1:2NAUJwPR47q+E35uaJeYoNhuNEM9kM8SjgRgdeOJUSE=
cloud.google.com/go v0.123.0/go.mod h1:xBoMV08QcqUGuPW65Qfm1o9Y4zKZBpGS+7bImXLTAZU=
cloud.google.com/go/auth v0.23.2 h1:pxSCpfiji41hpzpPdMCftEUCezpgpqmmDdYiAjCKXxo=
cloud.google.com/go/auth v0.23.2/go.mod h1:4DhBRcqvtljQN3dJ57qtqbib5ZGCYE5f2crfiiC2EM0=
cloud.google.com/go/auth v0.22.0 h1:Xp9wAKkLoeaYb5pYZZoQGz4E9sdPxIbzS3gywZE3ciQ=
cloud.google.com/go/auth v0.22.0/go.mod h1:M9o2Oz+YI2jAfxewJgb1vyI3vceHF+eohmxyzmrl+9s=
cloud.google.com/go/auth/oauth2adapt v0.2.8 h1:keo8NaayQZ6wimpNSmW5OPc283g65QNIiLpZnkHRbnc=
cloud.google.com/go/auth/oauth2adapt v0.2.8/go.mod h1:XQ9y31RkqZCcwJWNSx2Xvric3RrU88hAYYbjDWYDL+c=
cloud.google.com/go/compute/metadata v0.9.0 h1:pDUj4QMoPejqq20dK0Pg2N4yG9zIkYGdBtwLoEkH9Zs=
cloud.google.com/go/compute/metadata v0.9.0/go.mod h1:E0bWwX5wTnLPedCKqk3pJmVgCBSM6qQI1yTBdEb3C10=
cloud.google.com/go/firestore v1.25.0 h1:yY3rQKyQXNhnhETdseNayF6W1p4x0bdg9ZYS4hKJfOw=
cloud.google.com/go/firestore v1.25.0/go.mod h1:0PU6hj+r/QlhB6BLsRX+Kt/SYefTXrpYrBeHbYaSis8=
cloud.google.com/go/iam v1.13.0 h1:ufT3FPT5rFFXu6UtLkNoxaOaV5EuA1dsSkmemCSTo6U=
cloud.google.com/go/iam v1.13.0/go.mod h1:gHXdDEiPDvqd1q1KwBDGQlgZY/BwY760zU2LhOZS5w0=
cloud.google.com/go/logging v1.19.1 h1:7SsLhyTDBDrJw+Ll6Ns3I2mByqHXvJUc3rGjSlwiWgU=
cloud.google.com/go/logging v1.19.1/go.mod h1:2IkQ/d8jVJqV2qW8ZUGUiMjdZG1gkLD2JReGbZ8isqg=
cloud.google.com/go/firestore v1.24.0 h1:x0Z3hrgjYgo2wI9whuBRQcNc2hYwzZDQy/7pkUXbXcs=
cloud.google.com/go/firestore v1.24.0/go.mod h1:5aojyjN4olKUnBZDCRWwM+NsdrrCX3t1qfyERZGOonM=
cloud.google.com/go/iam v1.12.0 h1:Aki3bX9aHUDKPHfnRJfDcTdVedvy6quGBQcTqx3DRXk=
cloud.google.com/go/iam v1.12.0/go.mod h1:FEZ4lXpADAC2AIpQY7LANNjjwyQ2jK439CI2VaD+sLY=
cloud.google.com/go/logging v1.19.0 h1:NCqhdVUg3wQ8Cobdf16FDSuTGi3+6+hdSBHrY5TsR6Q=
cloud.google.com/go/logging v1.19.0/go.mod h1:i40NZCHC9Gqvod4yE+yQfDWwlgwW/SrshkkGibCHxcA=
cloud.google.com/go/longrunning v1.2.0 h1:WjYH3YHBGCxGJP9M4dWGHBfXr/cFIjMkNgWcJj7/iMM=
cloud.google.com/go/longrunning v1.2.0/go.mod h1:5KMQALFGOCtFoi2xSOA1u3H7WKlhmckgiyFw7+LGQp0=
cloud.google.com/go/monitoring v1.30.0 h1:r/d+JUbyKmJ8b07iznuKfzVzrIXTWxHQ3lBRm3x2LlY=
cloud.google.com/go/monitoring v1.30.0/go.mod h1:htlUR0QWVMrjFzZmN4LGnMAve9xB/eduwjmINxVZ8RM=
cloud.google.com/go/storage v1.65.1 h1:LRRpBJUTf+OXDPX9jZUKZ3mSLIsz3htG+qUpeNZovyA=
cloud.google.com/go/storage v1.65.1/go.mod h1:UsS9OgFg/XHOSYakQ8ZtLWWeyGkk1WnmD/GsGfN0BHM=
cloud.google.com/go/storage v1.64.0 h1:KLpxI/oX9LxeRsNqn877d2WyeT3ryiEwnGt8pwcSPZg=
cloud.google.com/go/storage v1.64.0/go.mod h1:lWyAtwvDZHdL3k68WVKbESP6bmWaV23ZJJ/JEVw/ZaQ=
cloud.google.com/go/trace v1.16.0 h1:GmQovzFc5F0CNfl0VLgL64aoTtu7xsM0YajW2GlG9+E=
cloud.google.com/go/trace v1.16.0/go.mod h1:r+bdAn16dKLSV1G2D5v3e58IlQlizfxWrUfjx7kM7X0=
firebase.google.com/go/v4 v4.21.0 h1:HBZV4jrLtFYj8EwWyqEZOuRLfkfkV2bpnfyyXHOhPxY=
@@ -28,14 +28,14 @@ github.com/AlekSi/pointer v1.2.0 h1:glcy/gc4h8HnG2Z3ZECSzZ1IX1x2JxRVuDzaJwQE0+w=
github.com/AlekSi/pointer v1.2.0/go.mod h1:gZGfd3dpW4vEc/UlyfKKi1roIqcCgwOIvb0tSNSBle0=
github.com/BurntSushi/toml v1.6.0 h1:dRaEfpa2VI55EwlIW72hMRHdWouJeRF7TPYhI+AUQjk=
github.com/BurntSushi/toml v1.6.0/go.mod h1:ukJfTF/6rtPPRCnwkur4qwRxa8vTRFBF0uk2lLoLwho=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.36.0 h1:3SdxXLkgAfiHRWcGTq6fneq9jgoJzneiY0yPQnjoT2E=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.36.0/go.mod h1:1iIdl0k+ppn9wT0wzR9H7HkSvIui/4qgtnKW10cQtds=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.60.0 h1:HldzheTs05E3ybqSitI/wHaof6+XERRudgZLjYbs3eE=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.60.0/go.mod h1:evkqaSczW9g2BQm1veCtgNhJ4wCCsRrOsSgNIn9LHQk=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/cloudmock v0.60.0 h1:Fx8NtDCmKH4ML2hUkPz4Dq250903vRDojMjVCDKwQuc=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/cloudmock v0.60.0/go.mod h1:V9g30lTKzfUsEW+gpWssck6u9IhARajmipodImLLcwI=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.60.0 h1:Oblia1QXBJlM/wOY9ARRUtsXdDYiMCzk3eCMikqoLbI=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.60.0/go.mod h1:SRAbhyZ4R4FagHMM9VtRgSY/lheRoht2fKelZXQUenk=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0 h1:bN1gA3of5bXtbnLsRPrwfmbbe7A5UWFlcTHseujLnpc=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0/go.mod h1:Yj5vHEz/aAepZGliRJsA6uvHAVAQyEwajq9ORCHPxzM=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.59.0 h1:c/Ivw7FuawPLfrr+zB0LZKeCchO2cAHQpF2qZ6OV7rQ=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.59.0/go.mod h1:Zba7lknY/d78oxbKqFTmCsaGwfpzeJ3ktrrLXtnTV6g=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/cloudmock v0.59.0 h1:xTXsqDOj5k9mK3VVWHYUryryJCIdYfXxdjKFwpzINUw=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/cloudmock v0.59.0/go.mod h1:V9g30lTKzfUsEW+gpWssck6u9IhARajmipodImLLcwI=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.59.0 h1:18FRm6ZcN/x9+ZmhMr96hLcTtlLn2/gHPuDLVeg7XcY=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.59.0/go.mod h1:YqwkQPrWSC7+byyc1VlKbWLBF5JsW5IoL6xUkemYSXk=
github.com/MicahParks/keyfunc v1.9.0 h1:lhKd5xrFHLNOWrDc4Tyb/Q1AJ4LCzQ48GVJyVIID3+o=
github.com/MicahParks/keyfunc v1.9.0/go.mod h1:IdnCilugA0O/99dW+/MkvlyrsX8+L8+x95xuVNtM5jw=
github.com/SherClockHolmes/webpush-go v1.4.0 h1:ocnzNKWN23T9nvHi6IfyrQjkIc0oJWv1B1pULsf9i3s=
@@ -50,8 +50,9 @@ github.com/cncf/xds/go v0.0.0-20260202195803-dba9d589def2 h1:aBangftG7EVZoUb69Os
github.com/cncf/xds/go v0.0.0-20260202195803-dba9d589def2/go.mod h1:qwXFYgsP6T7XnJtbKlf1HP8AjxZZyzxMmc+Lq5GjlU4=
github.com/cpuguy83/go-md2man/v2 v2.0.7 h1:zbFlGlXEAKlwXpmvle3d8Oe3YnkKIK4xSRTd3sHPnBo=
github.com/cpuguy83/go-md2man/v2 v2.0.7/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
github.com/creack/pty v1.1.9/go.mod h1:oKZEueFk5CKHvIhNR5MUki03XCEU+Q6VDXinZuGJ33E=
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
github.com/emersion/go-sasl v0.0.0-20200509203442-7bfe0ed36a21/go.mod h1:iL2twTeMvZnrg54ZoPDNfJaJaqy0xIQFuBdrLsmspwQ=
github.com/emersion/go-sasl v0.0.0-20241020182733-b788ff22d5a6 h1:oP4q0fw+fOSWn3DfFi4EXdT+B+gTtzx8GC9xsc26Znk=
github.com/emersion/go-sasl v0.0.0-20241020182733-b788ff22d5a6/go.mod h1:iL2twTeMvZnrg54ZoPDNfJaJaqy0xIQFuBdrLsmspwQ=
@@ -59,8 +60,8 @@ github.com/emersion/go-smtp v0.17.0 h1:tq90evlrcyqRfE6DSXaWVH54oX6OuZOQECEmhWBME
github.com/emersion/go-smtp v0.17.0/go.mod h1:qm27SGYgoIPRot6ubfQ/GpiPy/g3PaZAVRxiO/sDUgQ=
github.com/envoyproxy/go-control-plane v0.14.0 h1:hbG2kr4RuFj222B6+7T83thSPqLjwBIfQawTkC++2HA=
github.com/envoyproxy/go-control-plane v0.14.0/go.mod h1:NcS5X47pLl/hfqxU70yPwL9ZMkUlwlKxtAohpi2wBEU=
github.com/envoyproxy/go-control-plane/envoy v1.39.0 h1:1uwRDYPYG8BIBU9Mj1sUAebNmlM6beu/ZKKweSLDxk8=
github.com/envoyproxy/go-control-plane/envoy v1.39.0/go.mod h1:5e4ylfTZO723MEEFsCpSW4ZEBWR8mwkEyXfwJBTCZ9c=
github.com/envoyproxy/go-control-plane/envoy v1.37.0 h1:u3riX6BoYRfF4Dr7dwSOroNfdSbEPe9Yyl09/B6wBrQ=
github.com/envoyproxy/go-control-plane/envoy v1.37.0/go.mod h1:DReE9MMrmecPy+YvQOAOHNYMALuowAnbjjEMkkWOi6A=
github.com/envoyproxy/go-control-plane/ratelimit v0.1.0 h1:/G9QYbddjL25KvtKTv3an9lx6VBE2cnb8wp1vEGNYGI=
github.com/envoyproxy/go-control-plane/ratelimit v0.1.0/go.mod h1:Wk+tMFAFbCXaJPzVVHnPgRKdUdwW/KdbRt94AzgRee4=
github.com/envoyproxy/protoc-gen-validate v1.3.3 h1:MVQghNeW+LZcmXe7SY1V36Z+WFMDjpqGAGacLe2T0ds=
@@ -95,10 +96,10 @@ github.com/google/s2a-go v0.1.9 h1:LGD7gtMgezd8a/Xak7mEWL0PjoTQFvpRudN895yqKW0=
github.com/google/s2a-go v0.1.9/go.mod h1:YA0Ei2ZQL3acow2O62kdp9UlnvMmU7kA6Eutn0dXayM=
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
github.com/googleapis/enterprise-certificate-proxy v0.3.21 h1:OFdQ3tnCX/zaQ0Cedur3D3z7kI6HiLX9g3TiAN4/DFU=
github.com/googleapis/enterprise-certificate-proxy v0.3.21/go.mod h1:L3D/IQExI6LqEjBdXcZQ1WluSgigQmSwBboFstVPM4w=
github.com/googleapis/gax-go/v2 v2.24.0 h1:myMaPYyF9MecEmvQqMqomIwn9t/4KCZN9qnwsS76wlg=
github.com/googleapis/gax-go/v2 v2.24.0/go.mod h1:IaTHBDd7NHxSCiu0vEs8pQZu4dGZrWwuSoxCnk16OFM=
github.com/googleapis/enterprise-certificate-proxy v0.3.19 h1:mMOE7DN2+p76/EdIrmAy9B9bH+yC4563vmnJ34QR8i4=
github.com/googleapis/enterprise-certificate-proxy v0.3.19/go.mod h1:rSEsBUemEBZEexP2y6jPp16LUmUbjmSbcPMQizR0o4k=
github.com/googleapis/gax-go/v2 v2.23.0 h1:Tchl7qkvE7Ip3y+ztvNufYFvkfqTe7NfLTYGIdJRLuE=
github.com/googleapis/gax-go/v2 v2.23.0/go.mod h1:rBQKOVJCdb8IFEzg+FCwlt1LP/xMDGuqUXhUG+XMXEg=
github.com/gorilla/css v1.0.1 h1:ntNaBIghp6JmvWnxbZKANoLyuXTPZ4cAMlo6RyhlbO8=
github.com/gorilla/css v1.0.1/go.mod h1:BvnYkspnSzMmwRK+b8/xgNPLiIuNZr6vbZBTPQ2A3b0=
github.com/gorilla/websocket v1.5.3 h1:saDtZ6Pbx/0u+bgYQ3q96pZgCzfhKXGPqt7kZ72aNNg=
@@ -119,8 +120,8 @@ github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc=
github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw=
github.com/mattn/go-sqlite3 v1.14.50 h1:dmdFvo1XG4MPzA4IkAmE9upVz/Nj31uRoM5+jC8hYbY=
github.com/mattn/go-sqlite3 v1.14.50/go.mod h1:6JTjA44L93a0QCyJef5YvlPoKXntQPjzWv5gtm9sB6w=
github.com/mattn/go-sqlite3 v1.14.49 h1:B8jBHC3xhxZgxztrgruTuLucebnULQnx4W7cF7SAE9w=
github.com/mattn/go-sqlite3 v1.14.49/go.mod h1:6JTjA44L93a0QCyJef5YvlPoKXntQPjzWv5gtm9sB6w=
github.com/microcosm-cc/bluemonday v1.0.27 h1:MpEUotklkwCSLeH+Qdx1VJgNqLlpY2KXwXFM08ygZfk=
github.com/microcosm-cc/bluemonday v1.0.27/go.mod h1:jFi9vgW+H7c3V0lb6nR74Ib/DIB5OBs92Dimizgw2cA=
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA=
@@ -132,6 +133,8 @@ github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINE
github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 h1:GFCKgmp0tecUJ0sJuv4pzYCqS9+RGSn52M3FUwPs+uo=
github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10/go.mod h1:t/avpk3KcrXxUnYOhZhMXJlSEyie6gQbtLq5NM3loB8=
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 h1:Jamvg5psRIccs7FGNTlIRMkT8wgtp5eCXdBlqhYGL6U=
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU=
github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE=
github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk=
@@ -147,12 +150,12 @@ github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQD
github.com/spiffe/go-spiffe/v2 v2.8.1 h1:eXZMLsu+3MLEPJyGJkolqtVrteZfQdUpOWj6LTiDl/E=
github.com/spiffe/go-spiffe/v2 v2.8.1/go.mod h1:47Q0Q9/AqGha8QLHp+kxpH4Wca7X7EnOtlIJy3mxZ3U=
github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
github.com/stretchr/objx v0.5.3 h1:jmXUvGomnU1o3W/V5h2VEradbpJDwGrzugQQvL0POH4=
github.com/stretchr/objx v0.5.3/go.mod h1:rDQraq+vQZU7Fde9LOZLr8Tax6zZvy4kuNKF+QYS+U0=
github.com/stretchr/objx v0.5.2 h1:xuMeJ0Sdp5ZMRXx/aWO6RZxdr3beISkG5/G/aIRr3pY=
github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/8L+MA=
github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI=
github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg=
github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE=
github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg=
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
github.com/stripe/stripe-go/v74 v74.30.0 h1:0Kf0KkeFnY7iRhOwvTerX0Ia1BRw+eV1CVJ51mGYAUY=
github.com/stripe/stripe-go/v74 v74.30.0/go.mod h1:f9L6LvaXa35ja7eyvP6GQswoaIPaBRvGAimAO+udbBw=
github.com/urfave/cli/v2 v2.27.7 h1:bH59vdhbjLv3LAvIu6gd0usJHgoTTPhCFib8qqOwXYU=
@@ -162,40 +165,38 @@ github.com/xrash/smetrics v0.0.0-20250705151800-55b8f293f342/go.mod h1:Ohn+xnUBi
github.com/yuin/goldmark v1.4.13/go.mod h1:6yULJ656Px+3vBD8DxQVa3kxgyrAnzto9xy5taEt/CY=
go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64=
go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y=
go.opentelemetry.io/contrib/detectors/gcp v1.46.0 h1:PI8dGkqDaQkwJ8kOopqMhDTbrnK3UIeG/RCHH4HErbo=
go.opentelemetry.io/contrib/detectors/gcp v1.46.0/go.mod h1:nsrN5c/sOLoY2vsPxN/rQ0V0nvGrWJCqcW4UXLtqNG8=
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.71.0 h1:B2h3uqicet1CT2N5TOFhS+Gq++9i0/CLmaxvhmhtP5s=
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.71.0/go.mod h1:dylvB+ZiiwMvsDij9O84Uy7SijLgHMX4mbkncds+4Sw=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0 h1:3g7B90UzBltIDKq1/5mrTGxTnOFDV0ICOhLoxiZ8jlg=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0/go.mod h1:Ef8SuTh59BT7+ofpDxN9z+yOlc4t2GjLmKDgYNJL/NU=
go.opentelemetry.io/otel v1.46.0 h1:FHt5/CDyVxi/8IM1CH7VE/rRgq3kLHa2mSTVMO8AWyc=
go.opentelemetry.io/otel v1.46.0/go.mod h1:Gj3SEScelsNC45tp4nSxRYlS+f5iez7W8XPMCt905kE=
go.opentelemetry.io/contrib/detectors/gcp v1.44.0 h1:NmLfL734pJhM0JKaYd2Y28+nY9dPRWYAAbxhRCrKXPw=
go.opentelemetry.io/contrib/detectors/gcp v1.44.0/go.mod h1:tNAsgd8avTGke1+MndXlU5Cru4PQ9Ai/cCNWQv/ZJ/s=
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0 h1:2yEATaop1/a1I4psnSLgWVPLWwCzkqWakgJy7xTDVy0=
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0/go.mod h1:D7J12YRapIekYyPWgGPlA/23pRmpSEZC5xJC/TTLI9U=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 h1:8tvICD4vSTOOsNrsI4Ljf6C+6UKvpTEH5XY3JMoyPoo=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0/go.mod h1:z9+yiacE0IHRqM4qFfkbt/JYlmYXgss8GY/jXoNuPJI=
go.opentelemetry.io/otel v1.44.0 h1:JjwHmHpA4iZ3wBxluu2fbbE7j4kqlE8jXyAyPXH7HqU=
go.opentelemetry.io/otel v1.44.0/go.mod h1:BMgjTHL9WPRlRjL2oZCBTL4whCGtXch2H4BhOPIAyYc=
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0 h1:hqxVTu/GtBF+vJ8d1fzW7fRxZFvgoDjWcxwwCaFDYpU=
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0/go.mod h1:z5fVEF4X5v0ESvlJqBrrFlBVoj5EQuefZpzsu7R+x5Q=
go.opentelemetry.io/otel/metric v1.46.0 h1:yBnkXvgV7AXFILZc5K6IZe/CBFF3OS7BJ8ov6/lj0K8=
go.opentelemetry.io/otel/metric v1.46.0/go.mod h1:iPmdWqifKUdzziPkvvzIJXITl56fQx2mGM/DHLB3/2o=
go.opentelemetry.io/otel/metric/x v0.68.0 h1:TA/cBT23D3MnxYPwHL7YFOdYGdx0A0v+s7Mzotpd1dU=
go.opentelemetry.io/otel/metric/x v0.68.0/go.mod h1:agudOmvWhwUTjgibWDzxD2PoWYnpw5Ht5jISYOD2Hd4=
go.opentelemetry.io/otel/sdk v1.46.0 h1:h5CNQQjEbuQXY/JfZtgt3i7HVFV3aHPO2OAwO2eTYPI=
go.opentelemetry.io/otel/sdk v1.46.0/go.mod h1:GAERFXFt5SYCEB+YiKUbMBeza6UaDH7GmGOZEfh2gSM=
go.opentelemetry.io/otel/sdk/metric v1.46.0 h1:0piZ26EG4RBfebb2jhDH6ERCYHoVWduc3kLgPCwSnSE=
go.opentelemetry.io/otel/sdk/metric v1.46.0/go.mod h1:I1PbKrdVc8Qu8HYVDNtqVIwLwjNrhsV/uFuxfwg8mO4=
go.opentelemetry.io/otel/trace v1.46.0 h1:OULy7ccdJnZtJ0UDYFOIGaCmiWzJ8Vi2G/Rsu60qs1c=
go.opentelemetry.io/otel/trace v1.46.0/go.mod h1:J7GAXweO77XSFkB/rmAqk9D6ihszhFjLU+d9WuUxDLI=
go.opentelemetry.io/otel/metric v1.44.0 h1:1w0gILTcHdr3YI+ixLyjemwrVnsMURbTZFrSYCdDdmc=
go.opentelemetry.io/otel/metric v1.44.0/go.mod h1:8O7hanEPBNgEMmybD3s2VBKcgWOCsA6tzHBPODAiquo=
go.opentelemetry.io/otel/metric/x v0.66.0 h1:YkCrx1zLOChi9ZcZ6euupOcsgzbVlec7D/xoEU1+cTA=
go.opentelemetry.io/otel/metric/x v0.66.0/go.mod h1:d1+BDj9t96do0/1LoU1ayfCv79ZgNE41qbhBvnMOBZk=
go.opentelemetry.io/otel/sdk v1.44.0 h1:nHYwb9lK+fJPU/dnT6s7W7Z8itMWyqrnVfbheVYrZ58=
go.opentelemetry.io/otel/sdk v1.44.0/go.mod h1:Osuydd3Se74nqjAKxid74N5eC+jfEqfTegHRnq58oK0=
go.opentelemetry.io/otel/sdk/metric v1.44.0 h1:3LlKgI+VjbVsjNRFZJZAJ30WjXC5VkNRks6si09iEfI=
go.opentelemetry.io/otel/sdk/metric v1.44.0/go.mod h1:5B5pMARnXxKhltooO4xUuCBorl65a4EpnTalObqOigA=
go.opentelemetry.io/otel/trace v1.44.0 h1:jxF5CsGYCe74MCRx2X4g7WsY/VBKRqqpNvXlX/6gtIk=
go.opentelemetry.io/otel/trace v1.44.0/go.mod h1:oLl1jrMQAVo6v3GAggN+1VH9VIz9iUSvW53sW1Q8PIE=
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE=
go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ=
go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ=
go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
golang.org/x/crypto v0.0.0-20210921155107-089bfa567519/go.mod h1:GvvjBRRGRdwPK5ydBHafDWAxML/pGHZbMvKqRZ5+Abc=
golang.org/x/crypto v0.13.0/go.mod h1:y6Z2r+Rw4iayiXXAIxJIDAJ1zMW4yaTpebo8fPOliYc=
golang.org/x/crypto v0.19.0/go.mod h1:Iy9bg/ha4yyC70EfRS8jz+B6ybOBKMaSxLj6P6oBDfU=
golang.org/x/crypto v0.23.0/go.mod h1:CKFgDieR+mRhux2Lsu27y0fO304Db0wZe70UKqHu0v8=
golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
golang.org/x/crypto v0.54.0 h1:YLIA59K4fiNzHzjnZt2tUJQjQtUWfWbeHBqKtk3eScw=
golang.org/x/crypto v0.54.0/go.mod h1:KWL8ny2AZdGR2cWmzeHrp2azQPGogOv+HeQaVEXC2dk=
golang.org/x/mod v0.6.0-dev.0.20220419223038-86c51ed26bb4/go.mod h1:jJ57K6gSWd91VN4djpZkiMVwK6gcyfeH4XE8wZrZaV4=
golang.org/x/mod v0.8.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
golang.org/x/mod v0.12.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
@@ -210,8 +211,8 @@ golang.org/x/net v0.10.0/go.mod h1:0qNGK6F8kojg2nk9dLZ2mShWaEBan6FAoqfSigmmuDg=
golang.org/x/net v0.15.0/go.mod h1:idbUs1IY1+zTqbi8yxTbhexhEEk5ur9LInksu6HrEpk=
golang.org/x/net v0.21.0/go.mod h1:bIjVDfnllIU7BJ2DNgfnXvpSvtn8VRwhlsaeUTyUS44=
golang.org/x/net v0.25.0/go.mod h1:JkAGAh7GEvH74S6FOH42FLoXpXbE/aqXSrIQjXgsiwM=
golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To=
golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU=
golang.org/x/net v0.57.0 h1:K5+3DljvIuDG9/Jv9rvyMywYNFCQ9RSUY6OOTTkT+tE=
golang.org/x/net v0.57.0/go.mod h1:KpXc8iv+r3XplLAG/f7Jsf9RPszJzdR0f58q9vGOuEU=
golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
@@ -235,8 +236,8 @@ golang.org/x/sys v0.12.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.17.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
golang.org/x/sys v0.20.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
golang.org/x/sys v0.28.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og=
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/telemetry v0.0.0-20240228155512-f48c80bd79b2/go.mod h1:TeRTkGYfJXctD9OcfyVLyj2J3IxLnKwHJR8f4D8a3YE=
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
@@ -259,8 +260,8 @@ golang.org/x/text v0.13.0/go.mod h1:TvPlkZtksWOMsz7fbANvkp4WM8x/WCo/om8BMLbz+aE=
golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
golang.org/x/text v0.15.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
golang.org/x/text v0.21.0/go.mod h1:4IBbMaMmOPCJ8SecivzSH54+73PCFmPWxNTLm+vZkEQ=
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M=
golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs=
golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY=
golang.org/x/time v0.15.0 h1:bbrp8t3bGUeFOx08pvsMYRTCVSMk89u4tKbNOZbp88U=
golang.org/x/time v0.15.0/go.mod h1:Y4YMaQmXwGQZoFaVFk4YpCt4FLQMYKZe9oeV/f4MSno=
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
@@ -273,22 +274,22 @@ golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8T
golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4=
gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E=
google.golang.org/api v0.294.0 h1:8gASjJxdtcIieB3OqbkLcF0FfbXVNqKtU5iozD1ssvA=
google.golang.org/api v0.294.0/go.mod h1:02qB8+Ox1ZFzcaKFMguy1nQLJmSIyvV6Ff4txJEXtl4=
google.golang.org/api v0.291.0 h1:wfPbbY+mr9c7wZLqqzrHJLft/q8iFKREd6IgTBUene0=
google.golang.org/api v0.291.0/go.mod h1:at7kwWbuonglBFEBoeMDAV1bguHqL3qf0BHFsv3coa0=
google.golang.org/appengine/v2 v2.0.6 h1:LvPZLGuchSBslPBp+LAhihBeGSiRh1myRoYK4NtuBIw=
google.golang.org/appengine/v2 v2.0.6/go.mod h1:WoEXGoXNfa0mLvaH5sV3ZSGXwVmy8yf7Z1JKf3J3wLI=
google.golang.org/genproto v0.0.0-20260825221802-da73d73af1c5 h1:jPP56YzdY899KJ5W7efXHt/CkjlVfAaoFOwdi/IEAFA=
google.golang.org/genproto v0.0.0-20260825221802-da73d73af1c5/go.mod h1:gutZdP0DwAHp4vu5WaXgEK7tjsJ77ZEqzlOFWGZGziE=
google.golang.org/genproto/googleapis/api v0.0.0-20260825221802-da73d73af1c5 h1:izFU9hz7aeLI/Mi1J0991ae+xcwRLr7hTqWnB/9aIIU=
google.golang.org/genproto/googleapis/api v0.0.0-20260825221802-da73d73af1c5/go.mod h1:3LhxRw4YYkf+ylAfgaY9JlVLFKhokkCV8duhLLe7+t0=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260825221802-da73d73af1c5 h1:1VUiZAXyC+zmiFYi+WLtBzr68Cj8wOofHjjrA/kkizc=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260825221802-da73d73af1c5/go.mod h1:DjtHYE8FKJLivXcBEjGwndXfIC23G0VpXiXKqG179uA=
google.golang.org/grpc v1.83.2 h1:EManeRomTObA0BU7I8vXgg/78uE5MJ9M8B39EX2WscU=
google.golang.org/grpc v1.83.2/go.mod h1:YPI1hK3kDked6iHvgX3tR0y+nX/qpMFKhPgFsokw1S8=
google.golang.org/genproto v0.0.0-20260803160001-6ac0973c030d h1:33JLrUF0lFT31667SAtJzZAjLewV0ew5Mizks4caz0A=
google.golang.org/genproto v0.0.0-20260803160001-6ac0973c030d/go.mod h1:I7vGRdTamb7ukERkgP9I+0e4p21O4ak3cM7ICA3krg8=
google.golang.org/genproto/googleapis/api v0.0.0-20260803160001-6ac0973c030d h1:FarXi840EJWSHYTN3ERkADbPWjl307+FGrA22KAVjjc=
google.golang.org/genproto/googleapis/api v0.0.0-20260803160001-6ac0973c030d/go.mod h1:K/+WGbmBY7aNW1HDw1fJnKYo10i0DkAX6pows00dLig=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260803160001-6ac0973c030d h1:IL4hdHzcUv2l/gcg98/Rj3FbtE6axwqslOW8SW0C+S0=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260803160001-6ac0973c030d/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8=
google.golang.org/grpc v1.83.0 h1:JeNZEKJFbQxArAMl+hiytHauacDNqJUllNfmIMmpqnQ=
google.golang.org/grpc v1.83.0/go.mod h1:kDyl6SKsiHKt0uylY5gtn5cEjkrIOhQOGDgIc4JGwzQ=
google.golang.org/protobuf v1.26.0-rc.1/go.mod h1:jlhhOSvTdKEhbULTjvd4ARK9grFBp09yW+WbY/TyQbw=
google.golang.org/protobuf v1.30.0/go.mod h1:HV8QOd/L58Z+nl8r43ehVNZIU/HEI6OcFqwMG9pJV4I=
google.golang.org/protobuf v1.36.12 h1:pJOKDDOyeXErUroCihFAd5LQuwXBSpVnKGrj5o/fwxc=
google.golang.org/protobuf v1.36.12/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE=
google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk=
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q=
+47 -61
View File
@@ -5,7 +5,6 @@ import (
"encoding/json"
"errors"
"net/netip"
"slices"
"strings"
"sync"
"time"
@@ -19,9 +18,6 @@ import (
const (
tagMessageCache = "message_cache"
schemaStore = "message" // Store name in the schema_version table (see db/schema)
// NoLimit reads a topic's cached messages without a size budget.
NoLimit = 0
)
var errNoRows = errors.New("no rows found")
@@ -39,6 +35,7 @@ type queries struct {
selectMessagesSinceIDScheduled string
selectMessagesLatest string
selectMessagesDue string
selectMessagesDueForUpdate string // Postgres-only: claims due rows via FOR UPDATE SKIP LOCKED; empty for SQLite/mem
deleteExpiredMessages string
updateMessagePublished string
selectMessagesCount string
@@ -190,29 +187,19 @@ func (c *Cache) addMessages(ms []*model.Message) error {
return nil
}
// Messages returns all cached messages for a topic, oldest first. Prefer MessagesCapped on
// request paths: an uncapped replay of a busy topic is as large as the topic's entire cache.
// Messages returns messages for a topic since the given marker, optionally including scheduled messages
func (c *Cache) Messages(topic string, since model.SinceMarker, scheduled bool) ([]*model.Message, error) {
messages, _, err := c.MessagesCapped(topic, since, scheduled, NoLimit)
return messages, err
}
// MessagesCapped returns cached messages for a topic, oldest first, keeping the newest messages
// that fit in maxBytes worth of Message.Size (0 = no budget). The bool reports whether older messages
// were dropped, so the caller can tell the client that what it got is incomplete.
func (c *Cache) MessagesCapped(topic string, since model.SinceMarker, scheduled bool, maxBytes int64) ([]*model.Message, bool, error) {
if since.IsNone() {
return make([]*model.Message, 0), false, nil
return make([]*model.Message, 0), nil
} else if since.IsLatest() {
messages, err := c.messagesLatest(topic)
return messages, false, err
return c.messagesLatest(topic)
} else if since.IsID() {
return c.messagesSinceID(topic, since, scheduled, maxBytes)
return c.messagesSinceID(topic, since, scheduled)
}
return c.messagesSinceTime(topic, since, scheduled, maxBytes)
return c.messagesSinceTime(topic, since, scheduled)
}
func (c *Cache) messagesSinceTime(topic string, since model.SinceMarker, scheduled bool, maxBytes int64) ([]*model.Message, bool, error) {
func (c *Cache) messagesSinceTime(topic string, since model.SinceMarker, scheduled bool) ([]*model.Message, error) {
var rows *sql.Rows
var err error
rdb := c.db.ReadOnly()
@@ -222,12 +209,12 @@ func (c *Cache) messagesSinceTime(topic string, since model.SinceMarker, schedul
rows, err = rdb.Query(c.queries.selectMessagesSinceTime, topic, since.Time().Unix())
}
if err != nil {
return nil, false, err
return nil, err
}
return readMessagesCapped(rows, maxBytes)
return readMessages(rows)
}
func (c *Cache) messagesSinceID(topic string, since model.SinceMarker, scheduled bool, maxBytes int64) ([]*model.Message, bool, error) {
func (c *Cache) messagesSinceID(topic string, since model.SinceMarker, scheduled bool) ([]*model.Message, error) {
var rows *sql.Rows
var err error
rdb := c.db.ReadOnly()
@@ -237,9 +224,9 @@ func (c *Cache) messagesSinceID(topic string, since model.SinceMarker, scheduled
rows, err = rdb.Query(c.queries.selectMessagesSinceID, topic, since.ID())
}
if err != nil {
return nil, false, err
return nil, err
}
return readMessagesCapped(rows, maxBytes)
return readMessages(rows)
}
func (c *Cache) messagesLatest(topic string) ([]*model.Message, error) {
@@ -252,6 +239,15 @@ func (c *Cache) messagesLatest(topic string) ([]*model.Message, error) {
// MessagesDue returns all messages that are due for publishing
func (c *Cache) MessagesDue() ([]*model.Message, error) {
// On Postgres (cluster mode), claim due rows atomically so that concurrent delayed senders
// on other nodes cannot pick up the same message. We SELECT ... FOR UPDATE SKIP LOCKED and
// mark the claimed rows published in the same transaction; each row is thus handed to exactly
// one node. We deliberately mark published at claim time (not after delivery) to keep the row
// lock short: holding a transaction open across Firebase/WebPush/email delivery would be far
// worse than the small at-most-once window if a node crashes between claim and delivery.
if c.queries.selectMessagesDueForUpdate != "" {
return c.claimMessagesDue()
}
rows, err := c.db.Query(c.queries.selectMessagesDue, time.Now().Unix())
if err != nil {
return nil, err
@@ -259,6 +255,32 @@ func (c *Cache) MessagesDue() ([]*model.Message, error) {
return readMessages(rows)
}
// claimMessagesDue is the Postgres claiming path for MessagesDue (see its comment).
func (c *Cache) claimMessagesDue() ([]*model.Message, error) {
tx, err := c.db.Begin()
if err != nil {
return nil, err
}
defer tx.Rollback()
rows, err := tx.Query(c.queries.selectMessagesDueForUpdate, time.Now().Unix())
if err != nil {
return nil, err
}
messages, err := readMessages(rows) // reads all rows and closes them
if err != nil {
return nil, err
}
for _, m := range messages {
if _, err := tx.Exec(c.queries.updateMessagePublished, m.ID); err != nil {
return nil, err
}
}
if err := tx.Commit(); err != nil {
return nil, err
}
return messages, nil
}
// DeleteExpiredMessages deletes up to `limit` expired messages in a single query
// and returns the number of deleted rows.
func (c *Cache) DeleteExpiredMessages(limit int) (int64, error) {
@@ -472,42 +494,6 @@ func (c *Cache) processMessageBatches() {
}
}
// readMessagesCapped reads a newest-first result set, keeping the newest messages that fit in
// maxBytes worth of Message.Size (0 = no budget), and reverses them into the oldest-first
// order callers expect. It stops scanning once the budget is spent rather than reading everything
// and trimming, so a replay of a huge topic never materializes the whole cache. The bool reports
// whether older messages were left behind.
func readMessagesCapped(rows *sql.Rows, maxBytes int64) ([]*model.Message, bool, error) {
defer rows.Close()
messages := make([]*model.Message, 0)
truncated := false
var total int64
for rows.Next() {
m, err := readMessage(rows)
if err != nil {
return nil, false, err
}
if maxBytes > 0 {
size := int64(m.Size())
// Always return at least one message, even if it alone exceeds the budget: an empty
// reply is less useful than an oversized one, and the per-field limits bound how big it gets.
if len(messages) > 0 && total+size > maxBytes {
truncated = true
break
}
total += size
}
messages = append(messages, m)
}
if !truncated {
if err := rows.Err(); err != nil {
return nil, false, err
}
}
slices.Reverse(messages)
return messages, truncated, nil
}
func readMessages(rows *sql.Rows) ([]*model.Message, error) {
defer rows.Close()
messages := make([]*model.Message, 0)
+12 -4
View File
@@ -25,13 +25,13 @@ const (
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user_id, content_type, encoding
FROM message
WHERE topic = $1 AND time >= $2 AND published = TRUE
ORDER BY time DESC, id DESC
ORDER BY time, id
`
postgresSelectMessagesSinceTimeIncludeScheduledQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user_id, content_type, encoding
FROM message
WHERE topic = $1 AND time >= $2
ORDER BY time DESC, id DESC
ORDER BY time, id
`
postgresSelectMessagesSinceIDQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user_id, content_type, encoding
@@ -39,14 +39,14 @@ const (
WHERE topic = $1
AND id > COALESCE((SELECT id FROM message WHERE mid = $2), 0)
AND published = TRUE
ORDER BY time DESC, id DESC
ORDER BY time, id
`
postgresSelectMessagesSinceIDIncludeScheduledQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user_id, content_type, encoding
FROM message
WHERE topic = $1
AND (id > COALESCE((SELECT id FROM message WHERE mid = $2), 0) OR published = FALSE)
ORDER BY time DESC, id DESC
ORDER BY time, id
`
postgresSelectMessagesLatestQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user_id, content_type, encoding
@@ -61,6 +61,13 @@ const (
WHERE time <= $1 AND published = FALSE
ORDER BY time, id
`
postgresSelectMessagesDueForUpdateQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user_id, content_type, encoding
FROM message
WHERE time <= $1 AND published = FALSE
ORDER BY time, id
FOR UPDATE SKIP LOCKED
`
postgresUpdateMessagePublishedQuery = `UPDATE message SET published = TRUE WHERE mid = $1`
postgresSelectMessagesCountQuery = `SELECT COUNT(*) FROM message`
postgresSelectTopicsQuery = `SELECT topic FROM message GROUP BY topic`
@@ -88,6 +95,7 @@ var postgresQueries = queries{
selectMessagesSinceIDScheduled: postgresSelectMessagesSinceIDIncludeScheduledQuery,
selectMessagesLatest: postgresSelectMessagesLatestQuery,
selectMessagesDue: postgresSelectMessagesDueQuery,
selectMessagesDueForUpdate: postgresSelectMessagesDueForUpdateQuery,
deleteExpiredMessages: postgresDeleteExpiredMessagesQuery,
updateMessagePublished: postgresUpdateMessagePublishedQuery,
selectMessagesCount: postgresSelectMessagesCountQuery,
+4 -4
View File
@@ -31,25 +31,25 @@ const (
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user, content_type, encoding
FROM messages
WHERE topic = ? AND time >= ? AND published = 1
ORDER BY time DESC, id DESC
ORDER BY time, id
`
sqliteSelectMessagesSinceTimeIncludeScheduledQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user, content_type, encoding
FROM messages
WHERE topic = ? AND time >= ?
ORDER BY time DESC, id DESC
ORDER BY time, id
`
sqliteSelectMessagesSinceIDQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user, content_type, encoding
FROM messages
WHERE topic = ? AND id > COALESCE((SELECT id FROM messages WHERE mid = ?), 0) AND published = 1
ORDER BY time DESC, id DESC
ORDER BY time, id
`
sqliteSelectMessagesSinceIDIncludeScheduledQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user, content_type, encoding
FROM messages
WHERE topic = ? AND (id > COALESCE((SELECT id FROM messages WHERE mid = ?), 0) OR published = 0)
ORDER BY time DESC, id DESC
ORDER BY time, id
`
sqliteSelectMessagesLatestQuery = `
SELECT mid, sequence_id, time, event, expires, topic, message, title, priority, tags, click, icon, actions, attachment_name, attachment_type, attachment_size, attachment_expires, attachment_url, sender, user, content_type, encoding
+38
View File
@@ -1,6 +1,7 @@
package message_test
import (
"fmt"
"net/netip"
"path/filepath"
"sync"
@@ -556,6 +557,43 @@ func TestStore_MarkPublished(t *testing.T) {
})
}
func TestStore_MessagesDue_ClaimExactlyOnce(t *testing.T) {
// Postgres-only: exercises "FOR UPDATE SKIP LOCKED" claiming so that concurrent delayed
// senders running on different cluster nodes never pick up (and deliver) the same due
// message twice. With the non-claiming implementation, every concurrent caller sees every
// due row, so this test fails; with claiming, each row is returned to exactly one caller.
s := newTestPostgresStore(t) // skips if NTFY_TEST_DATABASE_URL is unset
const n = 40
for i := 0; i < n; i++ {
m := model.NewDefaultMessage("mytopic", fmt.Sprintf("scheduled %d", i))
m.Time = time.Now().Add(time.Hour).Unix() // future -> stored as published=FALSE
require.Nil(t, s.AddMessage(m))
// Move the time into the past so the message is due now (but still unpublished)
require.Nil(t, s.UpdateMessageTime(m.ID, time.Now().Add(-time.Minute).Unix()))
}
var mu sync.Mutex
seen := make(map[string]int)
var wg sync.WaitGroup
for c := 0; c < 6; c++ {
wg.Add(1)
go func() {
defer wg.Done()
due, err := s.MessagesDue()
require.Nil(t, err)
mu.Lock()
defer mu.Unlock()
for _, m := range due {
seen[m.ID]++
}
}()
}
wg.Wait()
require.Len(t, seen, n) // every due message was claimed
for id, count := range seen {
require.Equalf(t, 1, count, "message %s was claimed %d times, want exactly 1", id, count)
}
}
func TestStore_ExpireMessages(t *testing.T) {
forEachBackend(t, func(t *testing.T, s *message.Cache) {
// Add messages to two topics
+36
View File
@@ -75,6 +75,33 @@ var (
HTTPRequests = prometheus.NewCounterVec(prometheus.CounterOpts{
Name: "ntfy_http_requests_total",
}, []string{"http_code", "ntfy_code", "http_method"})
ClusterPeers = prometheus.NewGauge(prometheus.GaugeOpts{
Name: "ntfy_cluster_peers",
})
ClusterMessagesForwarded = prometheus.NewCounter(prometheus.CounterOpts{
Name: "ntfy_cluster_messages_forwarded_total",
})
ClusterSendErrors = prometheus.NewCounter(prometheus.CounterOpts{
Name: "ntfy_cluster_send_errors_total",
})
ClusterQueueDropped = prometheus.NewCounter(prometheus.CounterOpts{
Name: "ntfy_cluster_queue_dropped_total",
})
ClusterBatchesSent = prometheus.NewCounter(prometheus.CounterOpts{
Name: "ntfy_cluster_batches_sent_total",
})
ClusterMessagesWasted = prometheus.NewCounter(prometheus.CounterOpts{
Name: "ntfy_cluster_messages_wasted_total",
})
ClusterRouteSkipped = prometheus.NewCounter(prometheus.CounterOpts{
Name: "ntfy_cluster_route_skipped_total",
})
ClusterStatePushes = prometheus.NewCounter(prometheus.CounterOpts{
Name: "ntfy_cluster_state_pushes_total",
})
ClusterLeader = prometheus.NewGauge(prometheus.GaugeOpts{
Name: "ntfy_cluster_leader",
})
)
// init registers all collectors with the default Prometheus registry. Registration is
@@ -103,5 +130,14 @@ func init() {
Subscribers,
Topics,
HTTPRequests,
ClusterPeers,
ClusterMessagesForwarded,
ClusterSendErrors,
ClusterQueueDropped,
ClusterBatchesSent,
ClusterMessagesWasted,
ClusterRouteSkipped,
ClusterStatePushes,
ClusterLeader,
)
}
+9
View File
@@ -15,6 +15,15 @@ var expectedMetricNames = []string{
"ntfy_attachments_total_size",
"ntfy_calls_made_failure",
"ntfy_calls_made_success",
"ntfy_cluster_batches_sent_total",
"ntfy_cluster_leader",
"ntfy_cluster_messages_forwarded_total",
"ntfy_cluster_messages_wasted_total",
"ntfy_cluster_peers",
"ntfy_cluster_queue_dropped_total",
"ntfy_cluster_route_skipped_total",
"ntfy_cluster_send_errors_total",
"ntfy_cluster_state_pushes_total",
"ntfy_emails_received_failure",
"ntfy_emails_received_success",
"ntfy_emails_sent_failure",
-30
View File
@@ -101,24 +101,6 @@ func (m *Message) ForJSON() *Message {
return m
}
// Size returns an approximate byte size of the variable-length, publisher-controlled parts of a
// message. It is used to budget cache replays, so it deliberately counts every field a publisher
// can grow rather than trying to match the exact wire size.
func (m *Message) Size() int {
size := len(m.ID) + len(m.SequenceID) + len(m.Event) + len(m.Topic) + len(m.Title) +
len(m.Message) + len(m.Click) + len(m.Icon) + len(m.ContentType) + len(m.Encoding) + len(m.PollID)
for _, tag := range m.Tags {
size += len(tag)
}
for _, action := range m.Actions {
size += action.Size()
}
if m.Attachment != nil {
size += len(m.Attachment.Name) + len(m.Attachment.Type) + len(m.Attachment.URL)
}
return size
}
// Attachment represents a file attachment on a message
type Attachment struct {
Name string `json:"name"`
@@ -143,18 +125,6 @@ type Action struct {
Value string `json:"value,omitempty"` // used in "copy" action
}
// Size returns an approximate byte size of an action's variable-length fields.
func (a *Action) Size() int {
size := len(a.ID) + len(a.Action) + len(a.Label) + len(a.URL) + len(a.Method) + len(a.Body) + len(a.Intent) + len(a.Value)
for key, value := range a.Headers {
size += len(key) + len(value)
}
for key, value := range a.Extras {
size += len(key) + len(value)
}
return size
}
// NewAction creates a new action with initialized maps
func NewAction() *Action {
return &Action{
+14 -14
View File
@@ -10,6 +10,8 @@ import (
"text/template"
"time"
"heckel.io/ntfy/v2/cluster"
"heckel.io/ntfy/v2/ban"
"heckel.io/ntfy/v2/user"
)
@@ -73,16 +75,6 @@ const (
DefaultAttachmentExpiryDuration = 3 * time.Hour
DefaultAttachmentOrphanGracePeriod = time.Hour // Don't delete orphaned objects younger than this to avoid races with in-flight uploads
// DefaultMessagePollSizeLimit caps what one cache replay returns per topic. It is a backstop
// against a single request materializing an entire topic cache, not a tunable: on ntfy.sh it
// would fire on 2 of ~98k cached topics. See docs/subscribe/api.md#replay-limits.
DefaultMessagePollSizeLimit = 10 * 1024 * 1024
// messageTitleSizeLimit and messageTagsSizeLimit cap two publisher-controlled fields that
// otherwise have no limit of their own. Sized off ntfy.sh's own cache: title p999 is 212 bytes
// (16 of ~3M messages exceed 1 KB), tags p999 is 244 (197 exceed 512).
messageTitleSizeLimit = 1024
messageTagsSizeLimit = 512
)
// Defines all per-visitor limits
@@ -129,8 +121,13 @@ type Config struct {
ListenUnixMode fs.FileMode
KeyFile string
CertFile string
DatabaseURL string // PostgreSQL connection string (e.g. "postgres://user:pass@host:5432/ntfy")
DatabaseReplicaURLs []string // PostgreSQL read replica connection strings
DatabaseURL string // PostgreSQL connection string (e.g. "postgres://user:pass@host:5432/ntfy")
DatabaseReplicaURLs []string // PostgreSQL read replica connection strings
ClusterNodeID string // Stable per-node identifier used to skip a node's own fan-out; required in cluster mode
ClusterListen string // ip:port the dedicated cluster fan-out listener binds to (private network, e.g. "10.0.0.5:2587")
ClusterAdvertiseURL string // Base URL peers use to reach this node's fan-out listener (defaults to "http://<cluster-listen>")
ClusterSecret string `hash:"-"` // Shared secret authenticating node-to-node fan-out requests
ClusterBatchLinger time.Duration // How long fan-out messages wait to form a batch per peer; 0 sends immediately
FirebaseKeyFile string
CacheFile string
CacheDuration time.Duration
@@ -184,7 +181,6 @@ type Config struct {
MessageDelayMin time.Duration
MessageDelayMax time.Duration
MessageSizeLimit int
MessagePollSizeLimit int64
TotalTopicLimit int
TotalAttachmentSizeLimit int64
VisitorSubscriptionLimit int
@@ -247,6 +243,11 @@ func NewConfig() *Config {
KeyFile: "",
CertFile: "",
DatabaseURL: "",
ClusterNodeID: "",
ClusterListen: "",
ClusterAdvertiseURL: "",
ClusterSecret: "",
ClusterBatchLinger: cluster.DefaultBatchLinger,
FirebaseKeyFile: "",
CacheFile: "",
CacheDuration: DefaultCacheDuration,
@@ -293,7 +294,6 @@ func NewConfig() *Config {
TwilioVerifyService: "",
TwilioCallFormat: nil,
MessageSizeLimit: DefaultMessageSizeLimit,
MessagePollSizeLimit: DefaultMessagePollSizeLimit,
MessageDelayMin: DefaultMessageDelayMin,
MessageDelayMax: DefaultMessageDelayMax,
TotalTopicLimit: DefaultTotalTopicLimit,
+1
View File
@@ -25,6 +25,7 @@ func TestConfig_HashExcludesSecrets(t *testing.T) {
conf2.UpstreamAccessToken = "tk_upstream"
conf2.WebPushPrivateKey = "web-push-private-key"
conf2.SMTPSenderPass = "hunter2"
conf2.ClusterSecret = "cluster-secret"
conf2.AuthUsers = []*user.User{{Name: "phil", Hash: "$2a$10$somebcrypthash"}}
conf2.AuthTokens = map[string][]*user.Token{"phil": {{Value: "tk_secrettoken"}}}
assert.Equal(t, conf1.Hash(), conf2.Hash())
-2
View File
@@ -149,8 +149,6 @@ var (
errHTTPBadRequestAnonymousEmailNotAllowed = &errHTTP{40053, http.StatusBadRequest, "invalid request: anonymous email sending is not allowed", "https://ntfy.sh/docs/publish/#e-mail-notifications", nil}
errHTTPBadRequestResetLinkInvalid = &errHTTP{40054, http.StatusBadRequest, "invalid request: password reset link invalid or expired", "", nil}
errHTTPBadRequestTemplateTooLarge = &errHTTP{40056, http.StatusBadRequest, "invalid request: template too large", "https://ntfy.sh/docs/publish/#message-templating", nil}
errHTTPBadRequestTitleTooLarge = &errHTTP{40057, http.StatusBadRequest, "invalid request: title is too large", "https://ntfy.sh/docs/publish/#limitations", nil}
errHTTPBadRequestTagsTooLarge = &errHTTP{40058, http.StatusBadRequest, "invalid request: tags are too large", "https://ntfy.sh/docs/publish/#limitations", nil}
errHTTPNotFound = &errHTTP{40401, http.StatusNotFound, "page not found", "", nil}
errHTTPUnauthorized = &errHTTP{40101, http.StatusUnauthorized, "unauthorized", "https://ntfy.sh/docs/publish/#authentication", nil}
errHTTPForbidden = &errHTTP{40301, http.StatusForbidden, "forbidden", "https://ntfy.sh/docs/publish/#authentication", nil}
+139 -50
View File
@@ -32,6 +32,7 @@ import (
"heckel.io/ntfy/v2/action"
"heckel.io/ntfy/v2/attachment"
"heckel.io/ntfy/v2/ban"
"heckel.io/ntfy/v2/cluster"
"heckel.io/ntfy/v2/db"
"heckel.io/ntfy/v2/db/pg"
"heckel.io/ntfy/v2/log"
@@ -72,6 +73,8 @@ type Server struct {
stripe stripeAPI // Stripe API, can be replaced with a mock
priceCache *util.LookupCache[map[string]int64] // Stripe price ID -> price as cents (USD implied!)
metricsHandler http.Handler // Handles /metrics if enable-metrics set, and listen-metrics-http not set
cluster cluster.Cluster // Fans messages out to peer cluster nodes (nop when not clustered)
httpClusterServer *http.Server // Dedicated private listener for node-to-node fan-out (cluster-listen)
closeChan chan bool
mu sync.RWMutex
}
@@ -321,9 +324,93 @@ func New(conf *Config) (*Server, error) {
stripe: stripe,
}
s.priceCache = util.NewLookupCache(s.fetchStripePrices, conf.StripePriceCacheDuration)
// Cross-node cluster; delivery of peer messages to local subscribers is injected
// as a callback, so the cluster package never depends on the server. Peers talk to each
// other only via the dedicated cluster listener, never via the public listeners.
advertiseURL := conf.ClusterAdvertiseURL
if advertiseURL == "" && conf.ClusterListen != "" {
advertiseURL = "http://" + conf.ClusterListen
}
s.cluster, err = cluster.New(&cluster.Config{
Enabled: conf.ClusterListen != "", // Setting cluster-listen implicitly enables clustering
NodeID: cluster.NodeID(conf.ClusterNodeID),
AdvertiseURL: advertiseURL,
Secret: conf.ClusterSecret,
BatchLinger: conf.ClusterBatchLinger,
MaxMessageBytes: int64(conf.MessageSizeLimit)*4 + 1024, // Envelope overhead over the raw message
}, pool, s.deliverFromBus, s.liveTopics)
if err != nil {
return nil, err
}
// The cluster routes messages by subscription knowledge: peers learn this node's live topics
// via periodic state pushes (liveTopics) and immediate announcements on a topic's first
// subscriber (the hook below; also set in topicsFromIDs for topics created later)
for _, t := range s.topics {
t.onFirstSubscriber = s.topicAnnouncer(t.ID)
}
return s, nil
}
// clusterHandler returns the handler served on the dedicated cluster listener (cluster-listen).
// It serves the internal peer API (owned and routed by the cluster itself, including auth) plus
// a health endpoint; the public listeners never expose these paths, so internal traffic cannot
// be reached from the outside even before any firewalling.
func (s *Server) clusterHandler() http.Handler {
mux := http.NewServeMux()
mux.HandleFunc(apiHealthPath, func(w http.ResponseWriter, _ *http.Request) {
w.Header().Set("Content-Type", "application/json")
if s.cluster.Healthy() {
io.WriteString(w, `{"healthy":true}`+"\n")
} else {
w.WriteHeader(http.StatusServiceUnavailable)
io.WriteString(w, `{"healthy":false}`+"\n")
}
})
mux.Handle("/", s.cluster)
return mux
}
// liveTopics returns the topics that currently have at least one subscriber, computed fresh on
// every call. Deliberately NOT the whole topics map: it also holds subscriber-less topics
// rebuilt from the message cache, which would gut the routing filter's selectivity.
func (s *Server) liveTopics() []string {
s.mu.RLock()
defer s.mu.RUnlock()
topics := make([]string, 0, len(s.topics))
for _, t := range s.topics {
if subscribers, _ := t.Stats(); subscribers > 0 {
topics = append(topics, t.ID)
}
}
return topics
}
// topicAnnouncer returns the first-subscriber hook for a topic: it tells peer nodes right away
// that this node now wants messages for it (see Cluster.BroadcastState).
func (s *Server) topicAnnouncer(id string) func() {
return func() {
s.cluster.BroadcastState(&cluster.State{AddedTopics: []string{id}})
}
}
// deliverFromBus delivers a message received from a peer node (via the cluster) to this
// node's local subscribers. It is the receive-side counterpart to Cluster.ForwardMessage: local
// delivery and all global side effects (Firebase, email, web push, upstream) already ran on the
// origin node, so this only publishes to the local topic, and never re-relays.
func (s *Server) deliverFromBus(m *model.Message) {
s.mu.RLock()
t, ok := s.topics[m.Topic]
s.mu.RUnlock()
if !ok {
metrics.ClusterMessagesWasted.Inc() // Relayed here needlessly: this node had no use for the message
return
}
v := s.visitor(m.Sender, nil)
if err := t.Publish(v, m); err != nil {
logvm(v, m).Err(err).Warn("Cluster: unable to deliver fan-out message to local subscribers")
}
}
func createMessageCache(conf *Config, pool *db.DB) (*message.Cache, error) {
if conf.CacheDuration == 0 {
return message.NewNopStore()
@@ -366,6 +453,9 @@ func (s *Server) Run() error {
if s.config.ProfileListenHTTP != "" {
listenStr += fmt.Sprintf(" %s[http/profile]", s.config.ProfileListenHTTP)
}
if s.config.ClusterListen != "" {
listenStr += fmt.Sprintf(" %s[http/cluster]", s.config.ClusterListen)
}
log.Tag(tagStartup).Info("Listening on%s, ntfy %s, log level is %s", listenStr, s.config.BuildVersion, log.CurrentLevel().String())
if log.IsFile() {
fmt.Fprintf(os.Stderr, "Listening on%s, ntfy %s\n", listenStr, s.config.BuildVersion)
@@ -420,6 +510,12 @@ func (s *Server) Run() error {
} else if s.config.EnableMetrics {
s.metricsHandler = promhttp.Handler()
}
if s.config.ClusterListen != "" {
s.httpClusterServer = &http.Server{Addr: s.config.ClusterListen, Handler: s.clusterHandler()}
go func() {
errChan <- s.httpClusterServer.ListenAndServe()
}()
}
if s.config.ProfileListenHTTP != "" {
profileMux := http.NewServeMux()
profileMux.HandleFunc("/debug/pprof/", pprof.Index)
@@ -465,6 +561,12 @@ func (s *Server) Stop() {
if s.attachment != nil {
s.attachment.Close()
}
if s.httpClusterServer != nil {
s.httpClusterServer.Close()
}
if s.cluster != nil {
s.cluster.Close()
}
s.closeDatabases()
if s.ban != nil {
s.ban.Close()
@@ -717,10 +819,13 @@ func (s *Server) handleTopicAuth(w http.ResponseWriter, _ *http.Request, _ *visi
}
func (s *Server) handleHealth(w http.ResponseWriter, _ *http.Request, _ *visitor) error {
response := &apiHealthResponse{
Healthy: true,
// Unhealthy = the registry heartbeat went stale and peers stopped forwarding to this
// node; 503 lets status-code LB checks pull it (checkers must fail open, see cluster.Cluster)
healthy := s.cluster.Healthy()
if !healthy {
w.WriteHeader(http.StatusServiceUnavailable)
}
return s.writeJSON(w, response)
return s.writeJSON(w, &apiHealthResponse{Healthy: healthy})
}
// handleMetrics returns Prometheus metrics. This endpoint is only called if enable-metrics is set,
@@ -820,9 +925,13 @@ func (s *Server) handleMatrixDiscovery(w http.ResponseWriter) error {
return writeMatrixDiscoveryResponse(w)
}
// dispatch delivers m to local subscribers and fires the requested side-effect targets. It is
// the single choke point through which every published message must pass; t may be nil when
// the topic has no local subscribers (delayed sender).
// dispatch delivers m to local subscribers, forwards it to peer cluster nodes, and fires the
// requested side-effect targets. It is the single choke point through which every published
// message must pass; t may be nil when the topic has no local subscribers (delayed sender).
//
// The side-effect targets (Firebase, email, calls, upstream, web push) are global, not per-node:
// they fire only here on the origin node, and never again when a peer node receives the message
// via the cluster (see deliverFromBus).
func (s *Server) dispatch(v *visitor, t *topic, m *model.Message, opts dispatchOpts) error {
// Deliver to local subscribers
if t != nil {
@@ -836,6 +945,10 @@ func (s *Server) dispatch(v *visitor, t *topic, m *model.Message, opts dispatchO
return err
}
}
// Forward to peer cluster nodes, whose subscribers do not show up in this node's topics map
if err := s.cluster.ForwardMessage(m); err != nil {
logvm(v, m).Err(err).Warn("Cluster: unable to forward message to peer nodes")
}
// Fire the requested side-effect targets
if s.firebaseClient != nil && opts.firebase {
go s.sendToFirebase(v, m)
@@ -1041,7 +1154,7 @@ func (s *Server) handleActionMessage(w http.ResponseWriter, r *http.Request, v *
m.Sender = v.IP()
m.User = v.MaybeUserID()
m.Expires = time.Unix(m.Time, 0).Add(v.Limits().MessageExpiryDuration).Unix()
// Publish to subscribers, Firebase (for Android clients), and web push endpoints
// Publish to subscribers, peer nodes, Firebase (for Android clients), and web push endpoints
if err := s.dispatch(v, t, m, dispatchOpts{firebase: true, webPush: true}); err != nil {
return err
}
@@ -1147,9 +1260,6 @@ func (s *Server) parsePublishParams(r *http.Request, m *model.Message) (cache bo
cache = readBoolParam(r, true, "x-cache", "cache")
firebase = readBoolParam(r, true, "x-firebase", "firebase")
m.Title = readParam(r, "x-title", "title", "t")
if len(m.Title) > messageTitleSizeLimit {
return false, false, "", "", "", false, "", errHTTPBadRequestTitleTooLarge
}
m.Click = readParam(r, "x-click", "click")
icon := readParam(r, "x-icon", "icon")
filename := readParam(r, "x-filename", "filename", "file", "f")
@@ -1216,14 +1326,6 @@ func (s *Server) parsePublishParams(r *http.Request, m *model.Message) (cache bo
priorityStr = "" // Clear since it's already parsed
}
m.Tags = readCommaSeparatedParam(r, "x-tags", "tags", "tag", "ta")
// Measured across all tags, not each one: a publisher can add arbitrarily many
tagsSize := 0
for _, tag := range m.Tags {
tagsSize += len(tag)
}
if tagsSize > messageTagsSizeLimit {
return false, false, "", "", "", false, "", errHTTPBadRequestTagsTooLarge
}
delayStr := readParam(r, "x-delay", "delay", "x-at", "at", "x-in", "in")
if delayStr != "" {
if !cache {
@@ -1435,10 +1537,6 @@ func (s *Server) handleSubscribeHTTP(w http.ResponseWriter, r *http.Request, v *
}
var wlock sync.Mutex
var closed bool
// Only messages replayed from the cache are charged against the visitor's daily bandwidth
// budget, the same one attachment traffic uses. This is set in the poll branch below, which
// returns before any Subscribe, so sub() is never called concurrently while it is true.
meterPollBandwidth := false
defer func() {
// This blocks until any in-flight sub() call finishes writing/flushing the response writer,
// then marks the connection as closed so future sub() calls are no-ops. This prevents a panic
@@ -1453,22 +1551,16 @@ func (s *Server) handleSubscribeHTTP(w http.ResponseWriter, r *http.Request, v *
if !filters.Pass(msg) {
return nil
}
encoded, err := encoder(msg)
m, err := encoder(msg)
if err != nil {
return err
}
// Charge the encoded length, i.e. what actually goes over the wire. Charge before writing,
// so an exhausted budget fails the first message and surfaces as a clean 429 with nothing
// written.
if meterPollBandwidth && !v.BandwidthAllowed(int64(len(encoded))) {
return errHTTPTooManyRequestsLimitAttachmentBandwidth
}
wlock.Lock()
defer wlock.Unlock()
if closed {
return nil
}
if _, err := w.Write([]byte(encoded)); err != nil {
if _, err := w.Write([]byte(m)); err != nil {
return err
}
if fl, ok := w.(http.Flusher); ok {
@@ -1485,8 +1577,7 @@ func (s *Server) handleSubscribeHTTP(w http.ResponseWriter, r *http.Request, v *
for _, t := range topics {
t.Keepalive()
}
meterPollBandwidth = true
return s.sendOldMessages(w, topics, since, scheduled, v, sub)
return s.sendOldMessages(topics, since, scheduled, v, sub)
}
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
@@ -1502,7 +1593,7 @@ func (s *Server) handleSubscribeHTTP(w http.ResponseWriter, r *http.Request, v *
if err := sub(v, model.NewOpenMessage(topicsStr)); err != nil { // Send out open message
return err
}
if err := s.sendOldMessages(w, topics, since, scheduled, v, sub); err != nil {
if err := s.sendOldMessages(topics, since, scheduled, v, sub); err != nil {
return err
}
for {
@@ -1637,7 +1728,7 @@ func (s *Server) handleSubscribeWS(w http.ResponseWriter, r *http.Request, v *vi
for _, t := range topics {
t.Keepalive()
}
return s.sendOldMessages(w, topics, since, scheduled, v, sub)
return s.sendOldMessages(topics, since, scheduled, v, sub)
}
subscriberIDs := make([]int, 0)
for _, t := range topics {
@@ -1651,7 +1742,7 @@ func (s *Server) handleSubscribeWS(w http.ResponseWriter, r *http.Request, v *vi
if err := sub(v, model.NewOpenMessage(topicsStr)); err != nil { // Send out open message
return err
}
if err := s.sendOldMessages(w, topics, since, scheduled, v, sub); err != nil {
if err := s.sendOldMessages(topics, since, scheduled, v, sub); err != nil {
return err
}
err = g.Wait()
@@ -1746,30 +1837,21 @@ func (s *Server) setRateVisitors(r *http.Request, v *visitor, rateTopics []*topi
// sendOldMessages selects old messages from the messageCache and calls sub for each of them. It uses since as the
// marker, returning only messages that are newer than the marker.
func (s *Server) sendOldMessages(w http.ResponseWriter, topics []*topic, since model.SinceMarker, scheduled bool, v *visitor, sub subscriber) error {
func (s *Server) sendOldMessages(topics []*topic, since model.SinceMarker, scheduled bool, v *visitor, sub subscriber) error {
if since.IsNone() {
return nil
}
messages := make([]*model.Message, 0)
truncated := false
for _, t := range topics {
topicMessages, topicTruncated, err := s.messageCache.MessagesCapped(t.ID, since, scheduled, s.config.MessagePollSizeLimit)
topicMessages, err := s.messageCache.Messages(t.ID, since, scheduled)
if err != nil {
return err
}
truncated = truncated || topicTruncated
messages = append(messages, topicMessages...)
}
// Stable: Time has second granularity, so a multi-topic replay has many equal keys. An unstable
// sort reorders them and a topic's own messages come back out of publish order (#1297).
sort.SliceStable(messages, func(i, j int) bool {
sort.Slice(messages, func(i, j int) bool {
return messages[i].Time < messages[j].Time
})
// Must be set before the first message is written, or the header is already on the wire. On the
// WebSocket path the response has been hijacked by then, so this is a no-op there.
if truncated {
w.Header().Set("X-Messages-Truncated", "1")
}
for _, m := range messages {
if err := sub(v, m); err != nil {
return err
@@ -1869,7 +1951,9 @@ func (s *Server) topicsFromIDs(v *visitor, ids ...string) ([]*topic, error) {
if v != nil && !v.TopicCreationAllowed() {
return nil, errHTTPTooManyRequestsLimitTopicCreation
}
s.topics[id] = newTopic(id)
t := newTopic(id)
t.onFirstSubscriber = s.topicAnnouncer(id)
s.topics[id] = t
}
topics = append(topics, s.topics[id])
}
@@ -1956,7 +2040,8 @@ func (s *Server) resetStats() {
for _, v := range s.visitors {
v.ResetStats()
}
if s.userManager != nil {
// The user database is shared; only the cluster leader resets it (always true single-node)
if s.userManager != nil && s.cluster.IsLeader() {
if err := s.userManager.ResetStats(); err != nil {
log.Tag(tagResetter).Warn("Failed to write to database: %s", err.Error())
}
@@ -1971,7 +2056,11 @@ func (s *Server) runFirebaseKeepaliver() {
for {
select {
case <-time.After(s.config.FirebaseKeepaliveInterval):
s.sendToFirebase(v, model.NewKeepaliveMessage(firebaseControlTopic))
// Leader only: every FCM keepalive wakes all subscribed phones, so a cluster must
// send it exactly once, not once per node (checked per tick to survive failover)
if s.cluster.IsLeader() {
s.sendToFirebase(v, model.NewKeepaliveMessage(firebaseControlTopic))
}
/*
FIXME: Disable iOS polling entirely for now due to thundering herd problem (see #677)
To solve this, we'd have to shard the iOS poll topics to spread out the polling evenly.
+29 -4
View File
@@ -61,6 +61,34 @@
#
# database-url: <connection-string>
# If "cluster-listen" is set, clustering is implicitly enabled: this node registers itself in
# the PostgreSQL node registry and fans published messages out to the other cluster nodes over
# HTTP, so subscribers connected to any node receive messages published to any other node.
# Requires "database-url" and "cluster-secret".
#
# - cluster-listen is the ip:port of the dedicated fan-out listener that peer nodes talk to.
# Bind it to a private network interface (e.g. "10.0.0.5:2587"); the public listeners never
# serve the fan-out endpoint.
# - cluster-node-id is a stable per-node identifier (e.g. the hostname); required.
# - cluster-advertise-url is the base URL peer nodes use to reach this node's fan-out listener;
# it defaults to "http://<cluster-listen>". It must be set explicitly if cluster-listen binds
# a wildcard address (e.g. ":2587").
# - cluster-secret authenticates node-to-node fan-out requests; it must be identical on all
# nodes.
# - cluster-batch-linger is how long fan-out messages may wait to form a batch per peer node,
# trading up to that much cross-node delivery latency for a bounded request rate between
# nodes. Set to 0 to send each message immediately.
#
# SECURITY: The fan-out endpoint injects messages into arbitrary topics. The shared secret
# protects it, and it is only served on the dedicated cluster listener -- but you should still
# make sure that listener is reachable only from the private network (firewall/VPC rules).
#
# cluster-listen: <ip:port>
# cluster-node-id: <hostname>
# cluster-advertise-url: "http://<cluster-listen>"
# cluster-secret: <secret>
# cluster-batch-linger: 500ms
# If "cache-file" is set, messages are cached in a local SQLite database instead of only in-memory.
# This allows for service restarts without losing messages in support of the since= parameter.
# Not required if "database-url" is set (messages are stored in PostgreSQL instead).
@@ -417,10 +445,7 @@
# Rate limiting: Attachment size and bandwidth limits per visitor:
# - visitor-attachment-total-size-limit is the total storage limit used for attachments per visitor
# - visitor-attachment-daily-bandwidth-limit is the total daily traffic limit per visitor. It covers
# attachment downloads/uploads AND messages replayed from the message cache by poll requests. A
# poll without a "since" cursor returns the topic's entire cache, so a busy topic can be re-read
# for many times its own size; charging it here caps what one visitor can pull per day.
# - visitor-attachment-daily-bandwidth-limit is the total daily attachment download/upload traffic limit per visitor
#
# visitor-attachment-total-size-limit: "100M"
# visitor-attachment-daily-bandwidth-limit: "500M"
+1
View File
@@ -985,6 +985,7 @@ func (s *Server) publishSyncEventForUser(v *visitor, u *user.User) error {
return err
}
m := model.NewDefaultMessage(syncTopic.ID, string(messageBytes))
// Dispatch so the sync event also reaches the user's devices connected to peer cluster nodes
if err := s.dispatch(v, syncTopic, m, dispatchOpts{}); err != nil {
return err
}
+353
View File
@@ -0,0 +1,353 @@
package server
import (
"database/sql"
"net"
"net/http"
"net/http/httptest"
"net/netip"
"strings"
"sync"
"testing"
"time"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/cluster"
dbtest "heckel.io/ntfy/v2/db/test"
"heckel.io/ntfy/v2/model"
"heckel.io/ntfy/v2/user"
)
// fakeCluster records relayed messages and topic announcements so tests can assert that every
// publish path passes through the cluster exactly once, and that subscription hooks fire.
type fakeCluster struct {
mu sync.Mutex
messages []*model.Message
announced []string
notLeader bool
notHealthy bool
}
func (b *fakeCluster) ForwardMessage(m *model.Message) error {
b.mu.Lock()
defer b.mu.Unlock()
b.messages = append(b.messages, m)
return nil
}
func (b *fakeCluster) ServeHTTP(_ http.ResponseWriter, _ *http.Request) {}
func (b *fakeCluster) BroadcastState(state *cluster.State) {
b.mu.Lock()
defer b.mu.Unlock()
b.announced = append(b.announced, state.AddedTopics...)
}
func (b *fakeCluster) Healthy() bool {
b.mu.Lock()
defer b.mu.Unlock()
return !b.notHealthy
}
func (b *fakeCluster) setHealthy(healthy bool) {
b.mu.Lock()
defer b.mu.Unlock()
b.notHealthy = !healthy
}
func (b *fakeCluster) IsLeader() bool {
b.mu.Lock()
defer b.mu.Unlock()
return !b.notLeader
}
func (b *fakeCluster) setLeader(leader bool) {
b.mu.Lock()
defer b.mu.Unlock()
b.notLeader = !leader
}
func (b *fakeCluster) Close() error { return nil }
func (b *fakeCluster) Messages() []*model.Message {
b.mu.Lock()
defer b.mu.Unlock()
return append([]*model.Message{}, b.messages...)
}
func (b *fakeCluster) Announced() []string {
b.mu.Lock()
defer b.mu.Unlock()
return append([]string{}, b.announced...)
}
func TestServer_Cluster_PublishForwardsOnce(t *testing.T) {
s := newTestServer(t, newTestConfig(t, ""))
b := &fakeCluster{}
s.cluster = b
response := request(t, s, "PUT", "/mytopic", "hi there", nil)
require.Equal(t, 200, response.Code)
messages := b.Messages()
require.Len(t, messages, 1)
require.Equal(t, "mytopic", messages[0].Topic)
require.Equal(t, "hi there", messages[0].Message)
}
func TestServer_Cluster_SyncEventForwards(t *testing.T) {
// Account sync events are delivered via the user's st_... sync topic; without relaying
// them, cross-device account sync silently breaks when a user's devices land on different
// cluster nodes.
s := newTestServer(t, newTestConfig(t, ""))
b := &fakeCluster{}
s.cluster = b
u := &user.User{ID: "u_abc", Name: "phil", SyncTopic: "st_1234"}
v := s.visitor(netip.MustParseAddr("1.2.3.4"), nil)
require.Nil(t, s.publishSyncEventForUser(v, u))
messages := b.Messages()
require.Len(t, messages, 1)
require.Equal(t, "st_1234", messages[0].Topic)
}
func TestServer_Cluster_DeliverNotOnPublicHandler(t *testing.T) {
// The fan-out endpoint lives only on the dedicated cluster listener; the public handler must
// not serve it, even with cluster mode on and a valid secret.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
conf := newTestConfig(t, schemaDSN)
conf.ClusterNodeID = "node-a"
conf.ClusterListen = "127.0.0.1:1" // Enables clustering; not bound since Run() is not called
conf.ClusterSecret = "s3cret"
conf.ClusterAdvertiseURL = "http://127.0.0.1:1"
s := newTestServer(t, conf)
topics, err := s.topicsFromIDs(nil, "mytopic")
require.Nil(t, err)
var mu sync.Mutex
var received []*model.Message
topics[0].Subscribe(func(_ *visitor, m *model.Message) error {
mu.Lock()
defer mu.Unlock()
received = append(received, m)
return nil
}, "", func() {})
// A valid fan-out request against the PUBLIC handler must not deliver
response := request(t, s, "POST", "/v1/internal/message",
`{"message":{"id":"x1","time":1,"event":"message","topic":"mytopic","message":"sneaky"}}`,
map[string]string{"X-Cluster-Secret": "s3cret", "X-Cluster-Origin": "node-b"})
require.Equal(t, 404, response.Code)
time.Sleep(250 * time.Millisecond) // Delivery is async; give a wrong implementation time to fail
mu.Lock()
require.Empty(t, received)
mu.Unlock()
// The same request against the cluster listener handler DOES deliver
rr := httptest.NewRecorder()
req, err := http.NewRequest("POST", "/v1/internal/message",
strings.NewReader(`{"message":{"id":"x2","time":1,"event":"message","topic":"mytopic","message":"legit"}}`))
require.Nil(t, err)
req.Header.Set("X-Cluster-Secret", "s3cret")
req.Header.Set("X-Cluster-Origin", "node-b")
s.clusterHandler().ServeHTTP(rr, req)
require.Equal(t, 200, rr.Code)
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return len(received) == 1
})
}
func TestServer_Cluster_EndToEnd(t *testing.T) {
// Two full servers sharing one Postgres schema: a message published to node A over HTTP must
// reach a subscriber connected to node B, via the node registry and the fan-out endpoint.
schemaDSN := dbtest.CreateTestPostgresSchema(t)
// Node B: create the listener first so its advertise URL is known before the server exists
listenerB, err := net.Listen("tcp", "127.0.0.1:0")
require.Nil(t, err)
confB := newTestConfig(t, schemaDSN)
confB.ClusterNodeID = "node-b"
confB.ClusterListen = listenerB.Addr().String() // Enables clustering; the test serves it below
confB.ClusterSecret = "s3cret"
confB.ClusterAdvertiseURL = "http://" + listenerB.Addr().String()
sB := newTestServer(t, confB)
srvB := &http.Server{Handler: sB.clusterHandler()}
go srvB.Serve(listenerB)
defer srvB.Close()
// Node A: publish-only in this test, so its advertise URL is never called
confA := newTestConfig(t, schemaDSN)
confA.ClusterNodeID = "node-a"
confA.ClusterListen = "127.0.0.1:1" // Enables clustering; not bound since Run() is not called
confA.ClusterSecret = "s3cret"
confA.ClusterAdvertiseURL = "http://127.0.0.1:1"
sA := newTestServer(t, confA)
// Subscribe on node B
topics, err := sB.topicsFromIDs(nil, "mytopic")
require.Nil(t, err)
var mu sync.Mutex
var received []*model.Message
topics[0].Subscribe(func(_ *visitor, m *model.Message) error {
mu.Lock()
defer mu.Unlock()
received = append(received, m)
return nil
}, "", func() {})
// Publish on node A
response := request(t, sA, "PUT", "/mytopic", "hello cluster", nil)
require.Equal(t, 200, response.Code)
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return len(received) == 1
})
mu.Lock()
defer mu.Unlock()
require.Equal(t, "hello cluster", received[0].Message)
}
func TestServer_Cluster_DeliverFromBus(t *testing.T) {
// deliverFromBus is the receive side of the broadcaster: a message that originated on a peer
// node must reach this node's local subscribers, but must NOT be re-broadcast (loop) nor
// re-trigger origin-only side effects.
s := newTestServer(t, newTestConfig(t, ""))
b := &fakeCluster{}
s.cluster = b
topics, err := s.topicsFromIDs(nil, "mytopic")
require.Nil(t, err)
var mu sync.Mutex
var received []*model.Message
topics[0].Subscribe(func(_ *visitor, m *model.Message) error {
mu.Lock()
defer mu.Unlock()
received = append(received, m)
return nil
}, "", func() {})
m := model.NewDefaultMessage("mytopic", "from peer")
m.Sender = netip.MustParseAddr("5.6.7.8")
s.deliverFromBus(m)
waitFor(t, func() bool {
mu.Lock()
defer mu.Unlock()
return len(received) == 1
})
require.Empty(t, b.Messages()) // Peer messages are never re-relayed
}
func TestServer_Cluster_FirstSubscriberAnnounces(t *testing.T) {
// A topic gaining its FIRST subscriber is announced to peers exactly once, so publishers on
// other nodes stop skipping this node for it without waiting for the next state push.
s := newTestServer(t, newTestConfig(t, ""))
b := &fakeCluster{}
s.cluster = b
topics, err := s.topicsFromIDs(nil, "mytopic")
require.Nil(t, err)
subscriber := func(_ *visitor, _ *model.Message) error { return nil }
topics[0].Subscribe(subscriber, "", func() {})
waitFor(t, func() bool {
return len(b.Announced()) == 1 && b.Announced()[0] == "mytopic"
})
// A second subscriber does not re-announce
topics[0].Subscribe(subscriber, "", func() {})
time.Sleep(250 * time.Millisecond)
require.Len(t, b.Announced(), 1)
}
func TestServer_Cluster_ManagerPrunesOnlyOnLeader(t *testing.T) {
c := newTestConfig(t, "")
s := newTestServer(t, c)
cl := &fakeCluster{notLeader: true}
s.cluster = cl
// Publish and expire a message
rr := request(t, s, "POST", "/mytopic", "hi", nil)
require.Equal(t, 200, rr.Code)
m := toMessage(t, rr.Body.String())
require.Nil(t, s.messageCache.ExpireMessages("mytopic"))
// A non-leader node leaves shared-database pruning to the leader
s.execManager()
_, err := s.messageCache.Message(m.ID)
require.Nil(t, err)
// Once this node is the leader, the same run prunes
cl.setLeader(true)
s.execManager()
_, err = s.messageCache.Message(m.ID)
require.Equal(t, model.ErrMessageNotFound, err)
}
func TestServer_Cluster_StatsResetOnlyOnLeader(t *testing.T) {
c := newTestConfigWithAuthFile(t, "")
s := newTestServer(t, c)
cl := &fakeCluster{notLeader: true}
s.cluster = cl
// An anonymous visitor with an in-memory message count
v := newVisitor(c, s.messageCache, s.userManager, netip.MustParseAddr("1.2.3.4"), nil)
require.True(t, v.MessageAllowed())
s.mu.Lock()
s.visitors["ip:1.2.3.4"] = v
s.mu.Unlock()
require.Equal(t, int64(1), v.Stats().Messages)
// A user with persisted stats in the (shared) user database
require.Nil(t, s.userManager.AddUser("phil", "phil1234", user.RoleUser, false))
authDB, err := sql.Open("sqlite3", c.AuthFile)
require.Nil(t, err)
defer authDB.Close()
_, err = authDB.Exec(`UPDATE user SET stats_messages = 5 WHERE user = 'phil'`)
require.Nil(t, err)
// A non-leader node resets its own in-memory visitor stats, but leaves the user database
// to the leader
s.resetStats()
require.Equal(t, int64(0), v.Stats().Messages)
u, err := s.userManager.User("phil")
require.Nil(t, err)
require.Equal(t, int64(5), u.Stats.Messages)
// The leader resets the user database too
cl.setLeader(true)
s.resetStats()
u, err = s.userManager.User("phil")
require.Nil(t, err)
require.Equal(t, int64(0), u.Stats.Messages)
}
func TestServer_Cluster_FirebaseKeepaliverOnlyOnLeader(t *testing.T) {
// Every FCM keepalive wakes all subscribed phones, so only the leader may send them;
// N nodes sending N keepalives would multiply the battery cost for every user
c := newTestConfig(t, "")
c.FirebaseKeepaliveInterval = 20 * time.Millisecond
s := newTestServer(t, c)
sender := newTestFirebaseSender(100)
s.firebaseClient = newFirebaseClient(sender, &testAuther{Allow: true})
cl := &fakeCluster{notLeader: true}
s.cluster = cl
s.closeChan = make(chan bool) // Closed by Stop() in the test cleanup
go s.runFirebaseKeepaliver()
// A non-leader node stays silent
time.Sleep(150 * time.Millisecond)
require.Empty(t, sender.Messages())
// The leader sends keepalives
cl.setLeader(true)
waitFor(t, func() bool { return len(sender.Messages()) > 0 })
}
func TestServer_Cluster_HealthReflectsCluster(t *testing.T) {
// A node whose registry heartbeat went stale no longer receives forwarded messages, so
// health checks must pull it from rotation (the fail-open policy lives in the checker)
s := newTestServer(t, newTestConfig(t, ""))
cl := &fakeCluster{}
s.cluster = cl
rr := request(t, s, "GET", "/v1/health", "", nil)
require.Equal(t, 200, rr.Code)
require.Contains(t, rr.Body.String(), `"healthy":true`)
cl.setHealthy(false)
rr = request(t, s, "GET", "/v1/health", "", nil)
require.Equal(t, 503, rr.Code)
require.Contains(t, rr.Body.String(), `"healthy":false`)
// The cluster listener's health endpoint reflects the same state
rr2 := httptest.NewRecorder()
req, err := http.NewRequest("GET", "/v1/health", nil)
require.Nil(t, err)
s.clusterHandler().ServeHTTP(rr2, req)
require.Equal(t, 503, rr2.Code)
}
+9 -5
View File
@@ -10,12 +10,16 @@ func (s *Server) execManager() {
// WARNING: Make sure to only selectively lock with the mutex, and be aware that this
// there is no mutex for the entire function.
// Prune all the things
// Prune all the things. In-memory state is pruned on every node; jobs touching shared
// databases (and the web push job, which also sends expiry-warning notifications) run on
// the cluster leader only. In a single-node setup, IsLeader is always true.
s.pruneVisitors()
s.pruneTokens()
s.pruneAttachments()
s.pruneMessages()
s.pruneAndNotifyWebPushSubscriptions()
if s.cluster.IsLeader() {
s.pruneTokens()
s.pruneAttachments()
s.pruneMessages()
s.pruneAndNotifyWebPushSubscriptions()
}
// Message count
messagesCached, err := s.messageCache.MessagesCount()
-196
View File
@@ -16,7 +16,6 @@ import (
"os"
"path/filepath"
"runtime/debug"
"strconv"
"strings"
"sync"
"sync/atomic"
@@ -2811,201 +2810,6 @@ func TestServer_PublishAttachmentBandwidthLimit(t *testing.T) {
})
}
func TestServer_PollOrderAcrossTopics(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
// Replaying several topics at once concatenates each topic's messages and then sorts the
// lot by Time, which has second granularity. That sort must not reorder messages that
// share a timestamp, or a topic's own messages come back out of publish order. See #1297.
//
// The messages have to straddle a second boundary: if every timestamp is identical the
// concatenation is already sorted and Go's pdqsort leaves it alone, hiding the bug.
s := newTestServer(t, newTestConfig(t, databaseURL))
const perBatch = 10
publish := func(batch int) {
for _, topic := range []string{"topicA", "topicB"} {
for i := 0; i < perBatch; i++ {
body := fmt.Sprintf("%s-%02d", topic, batch*perBatch+i)
require.Equal(t, 200, request(t, s, "PUT", "/"+topic, body, nil).Code)
}
}
}
publish(0)
time.Sleep(1100 * time.Millisecond) // Cross a second boundary, so Time is not all-equal
publish(1)
response := request(t, s, "GET", "/topicA,topicB/json?poll=1", "", nil)
require.Equal(t, 200, response.Code)
messages := toMessages(t, response.Body.String())
require.Equal(t, 4*perBatch, len(messages))
// Each topic's own messages must appear in publish order, whatever the interleaving
lastSeen := map[string]int{"topicA": -1, "topicB": -1}
for _, m := range messages {
topic, seqStr, found := strings.Cut(m.Message, "-")
require.True(t, found)
seq, err := strconv.Atoi(seqStr)
require.Nil(t, err)
require.Greater(t, seq, lastSeen[topic], "%s came back out of publish order", m.Message)
lastSeen[topic] = seq
}
})
}
func TestServer_PublishTitleTooLarge(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
// Title has no length limit of its own, unlike the body, so it is capped here. Prod p999
// is 212 bytes and only 16 of ~3M cached messages exceed 1 KB.
s := newTestServer(t, newTestConfig(t, databaseURL))
require.Equal(t, 200, request(t, s, "PUT", "/mytopic", "x", map[string]string{
"Title": strings.Repeat("t", messageTitleSizeLimit),
}).Code)
response := request(t, s, "PUT", "/mytopic", "x", map[string]string{
"Title": strings.Repeat("t", messageTitleSizeLimit+1),
})
require.Equal(t, 400, response.Code)
require.Equal(t, 40057, toHTTPError(t, response.Body.String()).Code)
})
}
func TestServer_PublishTagsTooLarge(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
// Same for tags, measured across all of them: prod p999 is 244 bytes and only 197 of ~3M
// cached messages exceed 512.
s := newTestServer(t, newTestConfig(t, databaseURL))
require.Equal(t, 200, request(t, s, "PUT", "/mytopic", "x", map[string]string{
"Tags": strings.Repeat("g", messageTagsSizeLimit),
}).Code)
response := request(t, s, "PUT", "/mytopic", "x", map[string]string{
"Tags": strings.Repeat("g", messageTagsSizeLimit+1),
})
require.Equal(t, 400, response.Code)
require.Equal(t, 40058, toHTTPError(t, response.Body.String()).Code)
})
}
func TestServer_PollSizeLimit(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
// A poll without "since" replays the entire cache, which is unbounded in size. The cap is
// a byte budget rather than a message count, because message sizes vary ~20x in practice:
// a count cap truncates cheap high-volume topics while barely touching the expensive
// large-message ones it is meant to catch. The newest messages are kept.
c := newTestConfig(t, databaseURL)
c.MessagePollSizeLimit = 3500 // Fits three 1000-byte messages, not four
s := newTestServer(t, c)
for i := 0; i < 6; i++ {
body := fmt.Sprintf("%04d%s", i, strings.Repeat("x", 996)) // 1000 bytes, ordered prefix
require.Equal(t, 200, request(t, s, "PUT", "/mytopic", body, nil).Code)
}
response := request(t, s, "GET", "/mytopic/json?poll=1", "", nil)
require.Equal(t, 200, response.Code)
require.Equal(t, "1", response.Header().Get("X-Messages-Truncated"))
messages := toMessages(t, response.Body.String())
require.Equal(t, 3, len(messages))
require.Equal(t, "0003", messages[0].Message[:4]) // newest three, oldest first
require.Equal(t, "0004", messages[1].Message[:4])
require.Equal(t, "0005", messages[2].Message[:4])
// A topic under the budget is served whole, with no truncation header
require.Equal(t, 200, request(t, s, "PUT", "/othertopic", "small", nil).Code)
response = request(t, s, "GET", "/othertopic/json?poll=1", "", nil)
require.Equal(t, 200, response.Code)
require.Empty(t, response.Header().Get("X-Messages-Truncated"))
require.Equal(t, 1, len(toMessages(t, response.Body.String())))
})
}
func TestServer_PollSizeLimitCountsTitle(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
// Title is user-controlled and has no length limit of its own, so it has to count against
// the replay budget too; otherwise a topic of title-heavy messages sails past the cap.
c := newTestConfig(t, databaseURL)
c.MessagePollSizeLimit = 1600 // Fits one 500-byte title + 500-byte body, not two
s := newTestServer(t, c)
for i := 0; i < 4; i++ {
body := fmt.Sprintf("%04d%s", i, strings.Repeat("b", 496)) // 500 bytes
title := fmt.Sprintf("%04d%s", i, strings.Repeat("t", 496)) // 500 bytes
require.Equal(t, 200, request(t, s, "PUT", "/mytopic", body, map[string]string{"Title": title}).Code)
}
response := request(t, s, "GET", "/mytopic/json?poll=1", "", nil)
require.Equal(t, 200, response.Code)
require.Equal(t, "1", response.Header().Get("X-Messages-Truncated"))
messages := toMessages(t, response.Body.String())
require.Equal(t, 1, len(messages)) // 3 if the title were not counted
require.Equal(t, "0003", messages[0].Message[:4])
})
}
func TestServer_PollSizeLimitCountsEveryField(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
// Every field a publisher can grow has to count against the replay budget, not just the
// body and title: tags, click, icon and actions are all user-controlled, so anything left
// out is a hole the budget can be walked through.
c := newTestConfig(t, databaseURL)
c.MessagePollSizeLimit = 900 // Two messages fit if only body+title count; one if all fields do
s := newTestServer(t, c)
tags := make([]string, 5)
for i := range tags {
tags[i] = strings.Repeat("g", 79) // 395 bytes of tags
}
for i := 0; i < 3; i++ {
require.Equal(t, 200, request(t, s, "PUT", "/mytopic", fmt.Sprintf("%04d%s", i, strings.Repeat("b", 196)), map[string]string{
"Title": strings.Repeat("t", 200),
"Tags": strings.Join(tags, ","),
"Click": "https://example.com/" + strings.Repeat("c", 180),
}).Code)
}
response := request(t, s, "GET", "/mytopic/json?poll=1", "", nil)
require.Equal(t, 200, response.Code)
require.Equal(t, "1", response.Header().Get("X-Messages-Truncated"))
messages := toMessages(t, response.Body.String())
require.Equal(t, 1, len(messages)) // 2 if only body+title were counted
require.Equal(t, "0002", messages[0].Message[:4])
})
}
func TestServer_PollBandwidthLimit(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
// A poll without "since" replays the entire cache, so a topic that is cheap to fill is
// expensive to read over and over. Replayed bytes are charged against the same daily
// budget as attachment traffic. One message per poll keeps the accounting coarse: any
// shortfall hits the very first message, so the request is rejected before anything is
// written rather than truncated mid-stream.
c := newTestConfig(t, databaseURL)
c.VisitorAttachmentDailyBandwidthLimit = 9000 // Enough for two replays of the ~4 KB topic below, not three
s := newTestServer(t, c)
require.Equal(t, 200, request(t, s, "PUT", "/mytopic", util.RandomString(4000), nil).Code)
// Two full replays fit in the budget
for i := 1; i <= 2; i++ {
response := request(t, s, "GET", "/mytopic/json?poll=1", "", nil)
require.Equal(t, 200, response.Code)
require.Equal(t, 1, len(toMessages(t, response.Body.String())))
}
// The third is rejected before a single byte is written
response := request(t, s, "GET", "/mytopic/json?poll=1", "", nil)
require.Equal(t, 429, response.Code)
require.Equal(t, 42905, toHTTPError(t, response.Body.String()).Code)
// A subscription that replays nothing is not charged against the budget
response = request(t, s, "GET", "/mytopic/json?poll=1&since=none", "", nil)
require.Equal(t, 200, response.Code)
require.Empty(t, strings.TrimSpace(response.Body.String()))
})
}
func TestServer_PublishAttachmentBandwidthLimitUploadOnly(t *testing.T) {
forEachBackend(t, func(t *testing.T, databaseURL string) {
content := util.RandomString(5000) // > 4096
+10 -5
View File
@@ -20,11 +20,12 @@ const (
// topic represents a channel to which subscribers can subscribe, and publishers
// can publish a message
type topic struct {
ID string
subscribers map[int]*topicSubscriber
rateVisitor *visitor
lastAccess time.Time
mu sync.RWMutex
ID string
subscribers map[int]*topicSubscriber
rateVisitor *visitor
lastAccess time.Time
onFirstSubscriber func() // Fired (async) when the subscriber count goes 0 -> 1; may be nil
mu sync.RWMutex
}
type topicSubscriber struct {
@@ -56,6 +57,10 @@ func (t *topic) Subscribe(s subscriber, userID string, cancel func()) (subscribe
break
}
}
if len(t.subscribers) == 0 && t.onFirstSubscriber != nil {
// Fired async so cluster announcements never run under the topic lock
go t.onFirstSubscriber()
}
t.subscribers[subscriberID] = &topicSubscriber{
userID: userID, // May be empty
subscriber: s,
+1 -1
View File
@@ -30,7 +30,7 @@ type publishMessage struct {
}
// dispatchOpts selects which delivery targets fire for a published message, beyond delivery
// to local subscribers (see Server.dispatch)
// to local subscribers and the cross-node forward, which always happen (see Server.dispatch)
type dispatchOpts struct {
firebase bool // Send to Firebase (if configured)
email string // Send an email to this address (if a mailer is configured)
+1 -1
View File
@@ -65,7 +65,7 @@ type visitor struct {
callsLimiter *util.FixedLimiter // Rate limiter for calls
subscriptionLimiter *util.FixedLimiter // Fixed limiter for active subscriptions (ongoing connections)
topicCreationLimiter *rate.Limiter // Rate limiter for inserting new topics into the in-memory topic map
bandwidthLimiter *util.RateLimiter // Limiter for attachment downloads and cached-message replay (polls)
bandwidthLimiter *util.RateLimiter // Limiter for attachment bandwidth downloads
accountLimiter *rate.Limiter // Rate limiter for account actions (signup, password-reset requests), may be nil
authLimiter *rate.Limiter // Limiter for incorrect login attempts, may be nil
firebase time.Time // Next allowed Firebase message
+1 -1
View File
@@ -1 +1 @@
go1.27.0
go1.26.5
+3 -5
View File
@@ -167,13 +167,11 @@ func (t *Template) Delims(left, right string) *Template {
}
// Funcs adds the elements of the argument map to the template's function map.
// Any function used in the template must be added before the template is
// parsed. Funcs may be called more than once, including after parsing (for
// example, after [Template.Clone]), to replace a function of the same name;
// the replacement is used when the template is executed.
// It must be called before the template is parsed.
// It panics if a value in the map is not a function with appropriate return
// type or if the name cannot be used syntactically as a function in a template.
// The return value is the template, so calls can be chained.
// It is legal to overwrite elements of the map. The return value is the template,
// so calls can be chained.
func (t *Template) Funcs(funcMap FuncMap) *Template {
t.init()
t.muFuncs.Lock()
+103
View File
@@ -0,0 +1,103 @@
package util
import (
"errors"
"hash/fnv"
"math"
)
// BloomFilter is a fixed-size probabilistic set: Contains may return false positives (bounded by
// the target rate the filter was sized for) but never false negatives. Elements cannot be
// removed; rebuild the filter from scratch instead.
type BloomFilter struct {
bits []uint64
k int // number of hash probes per element, derived via double hashing
}
// NewBloomFilter creates a filter sized for n elements at the given false-positive rate.
func NewBloomFilter(n int, fpRate float64) *BloomFilter {
if n < 1 {
n = 1
}
if fpRate <= 0 || fpRate >= 1 {
fpRate = 0.01
}
m := int(math.Ceil(-float64(n) * math.Log(fpRate) / (math.Ln2 * math.Ln2))) // bits
k := int(math.Round(float64(m) / float64(n) * math.Ln2)) // probes
if k < 1 {
k = 1
}
return &BloomFilter{
bits: make([]uint64, (m+63)/64),
k: k,
}
}
// Add inserts an element into the filter.
func (b *BloomFilter) Add(s string) {
h1, h2 := hashPair(s)
m := uint64(len(b.bits)) * 64
for i := 0; i < b.k; i++ {
bit := (h1 + uint64(i)*h2) % m
b.bits[bit/64] |= 1 << (bit % 64)
}
}
// Contains reports whether the element may be in the set. A false result is definitive: the
// element was never added.
func (b *BloomFilter) Contains(s string) bool {
h1, h2 := hashPair(s)
m := uint64(len(b.bits)) * 64
for i := 0; i < b.k; i++ {
bit := (h1 + uint64(i)*h2) % m
if b.bits[bit/64]&(1<<(bit%64)) == 0 {
return false
}
}
return true
}
// MarshalBinary serializes the filter as [k][bits...], with 64-bit little-endian words.
func (b *BloomFilter) MarshalBinary() ([]byte, error) {
data := make([]byte, 1+len(b.bits)*8)
data[0] = byte(b.k)
for i, word := range b.bits {
for j := 0; j < 8; j++ {
data[1+i*8+j] = byte(word >> (8 * j))
}
}
return data, nil
}
// UnmarshalBloomFilter deserializes a filter produced by MarshalBinary.
func UnmarshalBloomFilter(data []byte) (*BloomFilter, error) {
if len(data) < 9 || (len(data)-1)%8 != 0 {
return nil, errors.New("invalid bloom filter data")
}
b := &BloomFilter{
bits: make([]uint64, (len(data)-1)/8),
k: int(data[0]),
}
if b.k < 1 {
return nil, errors.New("invalid bloom filter hash count")
}
for i := range b.bits {
var word uint64
for j := 0; j < 8; j++ {
word |= uint64(data[1+i*8+j]) << (8 * j)
}
b.bits[i] = word
}
return b, nil
}
// hashPair derives the two independent hash values used for double hashing (probe i uses
// h1 + i*h2), from FNV-1a over the element and a domain-separated variant of it.
func hashPair(s string) (uint64, uint64) {
f := fnv.New64a()
f.Write([]byte(s))
h1 := f.Sum64()
f.Write([]byte{0xff}) // Domain separation for the second hash
h2 := f.Sum64() | 1 // Odd, so probes cycle through all bit positions
return h1, h2
}
+61
View File
@@ -0,0 +1,61 @@
package util_test
import (
"fmt"
"testing"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/util"
)
func TestBloomFilter_NoFalseNegatives(t *testing.T) {
// The property routing correctness relies on: an added element is ALWAYS reported present
b := util.NewBloomFilter(1000, 0.01)
for i := 0; i < 1000; i++ {
b.Add(fmt.Sprintf("topic-%d", i))
}
for i := 0; i < 1000; i++ {
require.True(t, b.Contains(fmt.Sprintf("topic-%d", i)))
}
}
func TestBloomFilter_FalsePositiveRate(t *testing.T) {
b := util.NewBloomFilter(10000, 0.01)
for i := 0; i < 10000; i++ {
b.Add(fmt.Sprintf("added-%d", i))
}
falsePositives := 0
for i := 0; i < 10000; i++ {
if b.Contains(fmt.Sprintf("absent-%d", i)) {
falsePositives++
}
}
require.Less(t, falsePositives, 300, "expected ~1%% false positives, got %d/10000", falsePositives)
}
func TestBloomFilter_EmptyContainsNothing(t *testing.T) {
b := util.NewBloomFilter(100, 0.01)
require.False(t, b.Contains("anything"))
}
func TestBloomFilter_MarshalRoundTrip(t *testing.T) {
b := util.NewBloomFilter(500, 0.01)
for i := 0; i < 500; i++ {
b.Add(fmt.Sprintf("topic-%d", i))
}
data, err := b.MarshalBinary()
require.Nil(t, err)
b2, err := util.UnmarshalBloomFilter(data)
require.Nil(t, err)
for i := 0; i < 500; i++ {
require.True(t, b2.Contains(fmt.Sprintf("topic-%d", i)))
}
require.False(t, b2.Contains("never-added-topic"))
}
func TestBloomFilter_UnmarshalGarbage(t *testing.T) {
_, err := util.UnmarshalBloomFilter([]byte{})
require.Error(t, err)
_, err = util.UnmarshalBloomFilter([]byte{1, 2, 3})
require.Error(t, err)
}
+144
View File
@@ -0,0 +1,144 @@
package util
import (
"sync"
"time"
)
// LingerQueue is a bounded, non-blocking batching queue: enqueued elements are emitted as
// batches once a linger window expires, a batch count cap is reached, or a batch size cap is
// reached, whichever comes first. Unlike BatchingQueue, producers never block: TryEnqueue drops
// (returns false) when the queue is full, and Close flushes the remainder and terminates the
// consumer channel, so per-entity queues can be created and destroyed dynamically.
//
// Example:
//
// q := NewLingerQueue[int](64, 10, 0, nil, 500*time.Millisecond)
// go func() {
// for batch := range q.Dequeue() {
// send(batch)
// }
// }()
// q.TryEnqueue(1)
// q.TryEnqueue(2) // emitted together as [1, 2] after <= 500ms
type LingerQueue[T any] struct {
in chan T
out chan []T
max int // max elements per batch
maxSize int // max cumulative size per batch; 0 = no size cap
size func(T) int // element size function; nil = no size cap
linger time.Duration // max time the first element of a batch waits; 0 = emit immediately
closed bool
mu sync.Mutex // Protects closed, and guards TryEnqueue's send against Close's close(in)
}
// NewLingerQueue creates a LingerQueue holding up to capacity queued elements, emitting batches
// of up to max elements or maxSize cumulative size (as measured by size; pass 0/nil for no size
// cap) after at most linger.
func NewLingerQueue[T any](capacity, max, maxSize int, size func(T) int, linger time.Duration) *LingerQueue[T] {
q := &LingerQueue[T]{
in: make(chan T, capacity),
out: make(chan []T),
max: max,
maxSize: maxSize,
size: size,
linger: linger,
}
go q.run()
return q
}
// TryEnqueue enqueues an element without blocking. It returns false if the queue is full or
// closed; the caller decides how to account for the drop.
func (q *LingerQueue[T]) TryEnqueue(t T) bool {
q.mu.Lock()
defer q.mu.Unlock()
if q.closed {
return false
}
select {
case q.in <- t:
return true
default:
return false
}
}
// Dequeue returns the channel emitting batches. It is closed after Close, once the remaining
// elements have been flushed.
func (q *LingerQueue[T]) Dequeue() <-chan []T {
return q.out
}
// Close stops the queue: remaining elements are flushed as final batches, then the Dequeue
// channel is closed. TryEnqueue returns false after Close. Close is idempotent.
func (q *LingerQueue[T]) Close() {
q.mu.Lock()
defer q.mu.Unlock()
if q.closed {
return
}
q.closed = true
close(q.in)
}
// run is the batching loop: it blocks for the first element of a batch, then collects more until
// the linger expires or a cap is hit, and emits the batch. It exits once the queue is closed and
// drained. Note that receiving from the closed in channel still yields the buffered remainder
// before reporting closed, which is what flushes on Close.
func (q *LingerQueue[T]) run() {
defer close(q.out)
for {
first, ok := <-q.in
if !ok {
return
}
batch := []T{first}
bytes := q.sizeOf(first)
var timeout <-chan time.Time
if q.linger > 0 {
timeout = time.After(q.linger)
}
closed := false
collect:
for len(batch) < q.max && (q.maxSize <= 0 || bytes < q.maxSize) {
if timeout == nil {
// Zero linger: greedily drain what is immediately available, never wait
select {
case t, ok := <-q.in:
if !ok {
closed = true
break collect
}
batch = append(batch, t)
bytes += q.sizeOf(t)
default:
break collect
}
} else {
select {
case t, ok := <-q.in:
if !ok {
closed = true
break collect
}
batch = append(batch, t)
bytes += q.sizeOf(t)
case <-timeout:
break collect
}
}
}
q.out <- batch
if closed {
return
}
}
}
func (q *LingerQueue[T]) sizeOf(t T) int {
if q.size == nil {
return 0
}
return q.size(t)
}
+90
View File
@@ -0,0 +1,90 @@
package util_test
import (
"testing"
"time"
"github.com/stretchr/testify/require"
"heckel.io/ntfy/v2/util"
)
func TestLingerQueue_LingerFlush(t *testing.T) {
// Items enqueued within the linger window are emitted as a single batch when it expires
q := util.NewLingerQueue[int](16, 100, 0, nil, 50*time.Millisecond)
defer q.Close()
require.True(t, q.TryEnqueue(1))
require.True(t, q.TryEnqueue(2))
require.True(t, q.TryEnqueue(3))
start := time.Now()
batch := <-q.Dequeue()
require.Equal(t, []int{1, 2, 3}, batch)
require.GreaterOrEqual(t, time.Since(start), 30*time.Millisecond) // Waited out the linger
}
func TestLingerQueue_MaxBatchFlush(t *testing.T) {
// Hitting the count cap flushes early, before the linger expires
q := util.NewLingerQueue[int](16, 5, 0, nil, time.Minute)
defer q.Close()
for i := 0; i < 12; i++ {
require.True(t, q.TryEnqueue(i))
}
require.Len(t, <-q.Dequeue(), 5)
require.Len(t, <-q.Dequeue(), 5)
q.Close() // Flushes the remainder
require.Len(t, <-q.Dequeue(), 2)
}
func TestLingerQueue_SizeCapFlush(t *testing.T) {
// Hitting the byte cap flushes early, before count cap or linger
q := util.NewLingerQueue(16, 100, 10, func(s string) int { return len(s) }, time.Minute)
defer q.Close()
require.True(t, q.TryEnqueue("aaaa"))
require.True(t, q.TryEnqueue("bbbb"))
require.True(t, q.TryEnqueue("cccc")) // 12 bytes >= 10 -> flush
batch := <-q.Dequeue()
require.Equal(t, []string{"aaaa", "bbbb", "cccc"}, batch)
}
func TestLingerQueue_TryEnqueueFull(t *testing.T) {
// A full queue drops (returns false) instead of blocking the producer
q := util.NewLingerQueue[int](1, 1, 0, nil, 0)
defer q.Close()
require.True(t, q.TryEnqueue(1)) // Taken by the batcher, blocks emitting (no consumer)
waitForCond(t, func() bool { return q.TryEnqueue(2) }) // Fills the buffer once slot frees
require.False(t, q.TryEnqueue(3)) // Buffer full, batcher blocked -> drop
}
func TestLingerQueue_CloseFlushesAndCloses(t *testing.T) {
q := util.NewLingerQueue[int](16, 100, 0, nil, time.Minute)
require.True(t, q.TryEnqueue(1))
require.True(t, q.TryEnqueue(2))
q.Close()
require.Equal(t, []int{1, 2}, <-q.Dequeue()) // Remainder flushed without waiting out the linger
_, ok := <-q.Dequeue()
require.False(t, ok) // Channel closed
require.False(t, q.TryEnqueue(3))
q.Close() // Idempotent
}
func TestLingerQueue_LingerZeroImmediate(t *testing.T) {
q := util.NewLingerQueue[int](16, 100, 0, nil, 0)
defer q.Close()
require.True(t, q.TryEnqueue(1))
select {
case batch := <-q.Dequeue():
require.Contains(t, batch, 1)
case <-time.After(time.Second):
t.Fatal("expected immediate flush with zero linger")
}
}
func waitForCond(t *testing.T, f func() bool) {
t.Helper()
for i := 0; i < 100; i++ {
if f() {
return
}
time.Sleep(10 * time.Millisecond)
}
t.Fatal("timed out waiting for condition")
}
+544 -440
View File
File diff suppressed because it is too large Load Diff
+3 -1
View File
@@ -13,7 +13,9 @@ import (
"heckel.io/ntfy/v2/webpush"
)
const testWebPushEndpoint = "https://updates.push.services.mozilla.com/wpush/v1/AAABBCCCDDEEEFFF"
const (
testWebPushEndpoint = "https://updates.push.services.mozilla.com/wpush/v1/AAABBCCCDDEEEFFF"
)
// Schema layout as written by ntfy releases before the db/schema framework; used to verify
// that existing databases open cleanly without an adoption step