chore: modify replicasync to handle NATS explicitly (#26666)

relates to GRU-69

Modifies replicasync to handle discovering NATS enabled primary replicas explicitly, and passing that info to the NATS Pubsub.

This PR adds a new deployment value to explicitly represent the host or IP that the replica can be reached on. It isn't wired up to the CLI, but piggybacks on the DERP config for now.

We learn the NATS port directly from NATS at runtime, and propagate it thru replicasync to learn all peers for clustering.
This commit is contained in:
Spike Curtis
2026-06-26 08:36:32 -04:00
committed by GitHub
parent 9f211ce5ae
commit 98e1ce133c
16 changed files with 219 additions and 132 deletions
+24 -10
View File
@@ -36,6 +36,7 @@ type Options struct {
RelayAddress string
RegionID int32
TLSConfig *tls.Config
ClusterHost string
}
// New registers the replica with the database and periodically updates to
@@ -77,8 +78,8 @@ func New(ctx context.Context, logger slog.Logger, db database.Store, ps pubsub.P
// #nosec G115 - Safe conversion for microseconds latency which is expected to be within int32 range
DatabaseLatency: int32(databaseLatency.Microseconds()),
Primary: true,
ClusterHost: "", // TODO
NATSPort: 0, // TODO
ClusterHost: options.ClusterHost,
NATSPort: 0, // set later via SetSelfNATSPort
})
if err != nil {
return nil, xerrors.Errorf("insert replica: %w", err)
@@ -329,8 +330,8 @@ func (m *Manager) syncReplicas(ctx context.Context) error {
// #nosec G115 - Safe conversion for microseconds latency which is expected to be within int32 range
DatabaseLatency: int32(databaseLatency.Microseconds()),
Primary: m.self.Primary,
ClusterHost: "", // TODO
NATSPort: 0, // TODO
ClusterHost: m.self.ClusterHost,
NATSPort: m.self.NATSPort,
})
if err != nil {
if !errors.Is(err, sql.ErrNoRows) {
@@ -350,8 +351,8 @@ func (m *Manager) syncReplicas(ctx context.Context) error {
// #nosec G115 - Safe conversion for microseconds latency which is expected to be within int32 range
DatabaseLatency: int32(databaseLatency.Microseconds()),
Primary: m.self.Primary,
ClusterHost: "", // TODO
NATSPort: 0, // TODO
ClusterHost: m.self.ClusterHost,
NATSPort: m.self.NATSPort,
})
if err != nil {
return xerrors.Errorf("update replica: %w", err)
@@ -420,14 +421,27 @@ func (m *Manager) AllPrimary() []database.Replica {
return replicas
}
func (m *Manager) PrimaryPeerAddresses() []string {
func (m *Manager) FetchNATSPeers() []string {
addresses := make([]string, 0, len(m.AllPrimary()))
for _, replica := range m.AllPrimary() {
addresses = append(addresses, replica.RelayAddress)
if replica.ClusterHost == "" || replica.NATSPort == 0 {
continue
}
natsAddr := fmt.Sprintf("nats://%s:%d", replica.ClusterHost, replica.NATSPort)
addresses = append(addresses, natsAddr)
}
return addresses
}
func (m *Manager) SetSelfNATSPort(port int32) {
m.mutex.Lock()
defer m.mutex.Unlock()
m.self.NATSPort = port
m.logger.Debug(context.Background(), "nats port updated", slog.F("port", port))
// We're not really in a rush here, since it will take some time for our peers to dial and establish connections
// to us. So, we're not going to trigger a synchronous update. We'll just wait for the periodic update ticker.
}
// InRegion returns every replica in the given DERP region excluding itself.
func (m *Manager) InRegion(regionID int32) []database.Replica {
m.mutex.Lock()
@@ -503,8 +517,8 @@ func (m *Manager) Close() error {
Error: m.self.Error,
DatabaseLatency: 0, // A stopped replica has no latency.
Primary: false, // A stopped replica cannot be primary.
ClusterHost: "", // TODO
NATSPort: 0, // TODO
ClusterHost: m.self.ClusterHost,
NATSPort: 0, // A stopped replica cannot cluster with NATS
})
if err != nil {
return xerrors.Errorf("update replica: %w", err)
+19 -8
View File
@@ -279,17 +279,19 @@ func TestReplica(t *testing.T) {
require.NoError(t, server.UpdateNow(ctx))
requireNoCallback(t, called)
})
t.Run("PrimaryPeerAddresses", func(t *testing.T) {
t.Run("FetchNATSPeers", func(t *testing.T) {
t.Parallel()
db, pubsub := dbtestutil.NewDB(t)
ctx := testutil.Context(t, testutil.WaitShort)
primary, err := db.InsertReplica(ctx, database.InsertReplicaParams{
_, err := db.InsertReplica(ctx, database.InsertReplicaParams{
ID: uuid.New(),
CreatedAt: dbtime.Now(),
StartedAt: dbtime.Now(),
UpdatedAt: dbtime.Now(),
RelayAddress: "nats://primary.example:6222",
RelayAddress: "https://primary-relay.example",
Primary: true,
ClusterHost: "primary.example",
NATSPort: 6222,
})
require.NoError(t, err)
_, err = db.InsertReplica(ctx, database.InsertReplicaParams{
@@ -297,7 +299,7 @@ func TestReplica(t *testing.T) {
CreatedAt: dbtime.Now(),
StartedAt: dbtime.Now(),
UpdatedAt: dbtime.Now(),
RelayAddress: "nats://proxy.example:6222",
RelayAddress: "https://proxy-relay.example",
Primary: false,
})
require.NoError(t, err)
@@ -310,15 +312,24 @@ func TestReplica(t *testing.T) {
})
require.NoError(t, err)
server, err := replicasync.New(ctx, testutil.Logger(t), db, pubsub, &replicasync.Options{
RelayAddress: "nats://self.example:6222",
RelayAddress: "https://self-relay.example",
ClusterHost: "self.example",
UpdateInterval: time.Hour, // we'll explicitly trigger this
})
require.NoError(t, err)
defer server.Close()
require.Contains(t, server.PrimaryPeerAddresses(), primary.RelayAddress)
require.ElementsMatch(t, []string{
"nats://primary.example:6222",
"nats://self.example:6222",
}, server.PrimaryPeerAddresses())
}, server.FetchNATSPeers())
server.SetSelfNATSPort(6223)
err = server.UpdateNow(ctx)
require.NoError(t, err)
require.ElementsMatch(t, []string{
"nats://primary.example:6222",
"nats://self.example:6223",
}, server.FetchNATSPeers())
})
t.Run("TwentyConcurrent", func(t *testing.T) {
// Ensures that twenty concurrent replicas can spawn and all