mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
feat: Add high availability for multiple replicas (#4555)
* feat: HA tailnet coordinator * fixup! feat: HA tailnet coordinator * fixup! feat: HA tailnet coordinator * remove printlns * close all connections on coordinator * impelement high availability feature * fixup! impelement high availability feature * fixup! impelement high availability feature * fixup! impelement high availability feature * fixup! impelement high availability feature * Add replicas * Add DERP meshing to arbitrary addresses * Move packages to highavailability folder * Move coordinator to high availability package * Add flags for HA * Rename to replicasync * Denest packages for replicas * Add test for multiple replicas * Fix coordination test * Add HA to the helm chart * Rename function pointer * Add warnings for HA * Add the ability to block endpoints * Add flag to disable P2P connections * Wow, I made the tests pass * Add replicas endpoint * Ensure close kills replica * Update sql * Add database latency to high availability * Pipe TLS to DERP mesh * Fix DERP mesh with TLS * Add tests for TLS * Fix replica sync TLS * Fix RootCA for replica meshing * Remove ID from replicasync * Fix getting certificates for meshing * Remove excessive locking * Fix linting * Store mesh key in the database * Fix replica key for tests * Fix types gen * Fix unlocking unlocked * Fix race in tests * Update enterprise/derpmesh/derpmesh.go Co-authored-by: Colin Adler <colin1adler@gmail.com> * Rename to syncReplicas * Reuse http client * Delete old replicas on a CRON * Fix race condition in connection tests * Fix linting * Fix nil type * Move pubsub to in-memory for twenty test * Add comment for configuration tweaking * Fix leak with transport * Fix close leak in derpmesh * Fix race when creating server * Remove handler update * Skip test on Windows * Fix DERP mesh test * Wrap HTTP handler replacement in mutex * Fix error message for relay * Fix API handler for normal tests * Fix speedtest * Fix replica resend * Fix derpmesh send * Ping async * Increase wait time of template version jobd * Fix race when closing replica sync * Add name to client * Log the derpmap being used * Don't connect if DERP is empty * Improve agent coordinator logging * Fix lock in coordinator * Fix relay addr * Fix race when updating durations * Fix client publish race * Run pubsub loop in a queue * Store agent nodes in order * Fix coordinator locking * Check for closed pipe Co-authored-by: Colin Adler <colin1adler@gmail.com>
This commit is contained in:
co-authored by
Colin Adler
parent
dc3519e973
commit
2ba4a62a0d
@@ -0,0 +1,575 @@
|
||||
package tailnet
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"net"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"golang.org/x/xerrors"
|
||||
|
||||
"cdr.dev/slog"
|
||||
"github.com/coder/coder/coderd/database"
|
||||
agpl "github.com/coder/coder/tailnet"
|
||||
)
|
||||
|
||||
// NewCoordinator creates a new high availability coordinator
|
||||
// that uses PostgreSQL pubsub to exchange handshakes.
|
||||
func NewCoordinator(logger slog.Logger, pubsub database.Pubsub) (agpl.Coordinator, error) {
|
||||
ctx, cancelFunc := context.WithCancel(context.Background())
|
||||
coord := &haCoordinator{
|
||||
id: uuid.New(),
|
||||
log: logger,
|
||||
pubsub: pubsub,
|
||||
closeFunc: cancelFunc,
|
||||
close: make(chan struct{}),
|
||||
nodes: map[uuid.UUID]*agpl.Node{},
|
||||
agentSockets: map[uuid.UUID]net.Conn{},
|
||||
agentToConnectionSockets: map[uuid.UUID]map[uuid.UUID]net.Conn{},
|
||||
}
|
||||
|
||||
if err := coord.runPubsub(ctx); err != nil {
|
||||
return nil, xerrors.Errorf("run coordinator pubsub: %w", err)
|
||||
}
|
||||
|
||||
return coord, nil
|
||||
}
|
||||
|
||||
type haCoordinator struct {
|
||||
id uuid.UUID
|
||||
log slog.Logger
|
||||
mutex sync.RWMutex
|
||||
pubsub database.Pubsub
|
||||
close chan struct{}
|
||||
closeFunc context.CancelFunc
|
||||
|
||||
// nodes maps agent and connection IDs their respective node.
|
||||
nodes map[uuid.UUID]*agpl.Node
|
||||
// agentSockets maps agent IDs to their open websocket.
|
||||
agentSockets map[uuid.UUID]net.Conn
|
||||
// agentToConnectionSockets maps agent IDs to connection IDs of conns that
|
||||
// are subscribed to updates for that agent.
|
||||
agentToConnectionSockets map[uuid.UUID]map[uuid.UUID]net.Conn
|
||||
}
|
||||
|
||||
// Node returns an in-memory node by ID.
|
||||
func (c *haCoordinator) Node(id uuid.UUID) *agpl.Node {
|
||||
c.mutex.Lock()
|
||||
defer c.mutex.Unlock()
|
||||
node := c.nodes[id]
|
||||
return node
|
||||
}
|
||||
|
||||
// ServeClient accepts a WebSocket connection that wants to connect to an agent
|
||||
// with the specified ID.
|
||||
func (c *haCoordinator) ServeClient(conn net.Conn, id uuid.UUID, agent uuid.UUID) error {
|
||||
c.mutex.Lock()
|
||||
// When a new connection is requested, we update it with the latest
|
||||
// node of the agent. This allows the connection to establish.
|
||||
node, ok := c.nodes[agent]
|
||||
c.mutex.Unlock()
|
||||
if ok {
|
||||
data, err := json.Marshal([]*agpl.Node{node})
|
||||
if err != nil {
|
||||
return xerrors.Errorf("marshal node: %w", err)
|
||||
}
|
||||
_, err = conn.Write(data)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("write nodes: %w", err)
|
||||
}
|
||||
} else {
|
||||
err := c.publishClientHello(agent)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish client hello: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
c.mutex.Lock()
|
||||
connectionSockets, ok := c.agentToConnectionSockets[agent]
|
||||
if !ok {
|
||||
connectionSockets = map[uuid.UUID]net.Conn{}
|
||||
c.agentToConnectionSockets[agent] = connectionSockets
|
||||
}
|
||||
|
||||
// Insert this connection into a map so the agent can publish node updates.
|
||||
connectionSockets[id] = conn
|
||||
c.mutex.Unlock()
|
||||
|
||||
defer func() {
|
||||
c.mutex.Lock()
|
||||
defer c.mutex.Unlock()
|
||||
// Clean all traces of this connection from the map.
|
||||
delete(c.nodes, id)
|
||||
connectionSockets, ok := c.agentToConnectionSockets[agent]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
delete(connectionSockets, id)
|
||||
if len(connectionSockets) != 0 {
|
||||
return
|
||||
}
|
||||
delete(c.agentToConnectionSockets, agent)
|
||||
}()
|
||||
|
||||
decoder := json.NewDecoder(conn)
|
||||
// Indefinitely handle messages from the client websocket.
|
||||
for {
|
||||
err := c.handleNextClientMessage(id, agent, decoder)
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) || errors.Is(err, io.ErrClosedPipe) {
|
||||
return nil
|
||||
}
|
||||
return xerrors.Errorf("handle next client message: %w", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (c *haCoordinator) handleNextClientMessage(id, agent uuid.UUID, decoder *json.Decoder) error {
|
||||
var node agpl.Node
|
||||
err := decoder.Decode(&node)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("read json: %w", err)
|
||||
}
|
||||
|
||||
c.mutex.Lock()
|
||||
// Update the node of this client in our in-memory map. If an agent entirely
|
||||
// shuts down and reconnects, it needs to be aware of all clients attempting
|
||||
// to establish connections.
|
||||
c.nodes[id] = &node
|
||||
// Write the new node from this client to the actively connected agent.
|
||||
agentSocket, ok := c.agentSockets[agent]
|
||||
c.mutex.Unlock()
|
||||
if !ok {
|
||||
// If we don't own the agent locally, send it over pubsub to a node that
|
||||
// owns the agent.
|
||||
err := c.publishNodesToAgent(agent, []*agpl.Node{&node})
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish node to agent")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Write the new node from this client to the actively
|
||||
// connected agent.
|
||||
data, err := json.Marshal([]*agpl.Node{&node})
|
||||
if err != nil {
|
||||
return xerrors.Errorf("marshal nodes: %w", err)
|
||||
}
|
||||
|
||||
_, err = agentSocket.Write(data)
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) || errors.Is(err, io.ErrClosedPipe) {
|
||||
return nil
|
||||
}
|
||||
return xerrors.Errorf("write json: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// ServeAgent accepts a WebSocket connection to an agent that listens to
|
||||
// incoming connections and publishes node updates.
|
||||
func (c *haCoordinator) ServeAgent(conn net.Conn, id uuid.UUID) error {
|
||||
// Tell clients on other instances to send a callmemaybe to us.
|
||||
err := c.publishAgentHello(id)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish agent hello: %w", err)
|
||||
}
|
||||
|
||||
// Publish all nodes on this instance that want to connect to this agent.
|
||||
nodes := c.nodesSubscribedToAgent(id)
|
||||
if len(nodes) > 0 {
|
||||
data, err := json.Marshal(nodes)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("marshal json: %w", err)
|
||||
}
|
||||
_, err = conn.Write(data)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("write nodes: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
// If an old agent socket is connected, we close it
|
||||
// to avoid any leaks. This shouldn't ever occur because
|
||||
// we expect one agent to be running.
|
||||
c.mutex.Lock()
|
||||
oldAgentSocket, ok := c.agentSockets[id]
|
||||
if ok {
|
||||
_ = oldAgentSocket.Close()
|
||||
}
|
||||
c.agentSockets[id] = conn
|
||||
c.mutex.Unlock()
|
||||
defer func() {
|
||||
c.mutex.Lock()
|
||||
defer c.mutex.Unlock()
|
||||
delete(c.agentSockets, id)
|
||||
delete(c.nodes, id)
|
||||
}()
|
||||
|
||||
decoder := json.NewDecoder(conn)
|
||||
for {
|
||||
node, err := c.handleAgentUpdate(id, decoder)
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) || errors.Is(err, io.ErrClosedPipe) {
|
||||
return nil
|
||||
}
|
||||
return xerrors.Errorf("handle next agent message: %w", err)
|
||||
}
|
||||
|
||||
err = c.publishAgentToNodes(id, node)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish agent to nodes: %w", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (c *haCoordinator) nodesSubscribedToAgent(agentID uuid.UUID) []*agpl.Node {
|
||||
c.mutex.Lock()
|
||||
defer c.mutex.Unlock()
|
||||
sockets, ok := c.agentToConnectionSockets[agentID]
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
|
||||
nodes := make([]*agpl.Node, 0, len(sockets))
|
||||
for targetID := range sockets {
|
||||
node, ok := c.nodes[targetID]
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
nodes = append(nodes, node)
|
||||
}
|
||||
|
||||
return nodes
|
||||
}
|
||||
|
||||
func (c *haCoordinator) handleClientHello(id uuid.UUID) error {
|
||||
c.mutex.Lock()
|
||||
node, ok := c.nodes[id]
|
||||
c.mutex.Unlock()
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
return c.publishAgentToNodes(id, node)
|
||||
}
|
||||
|
||||
func (c *haCoordinator) handleAgentUpdate(id uuid.UUID, decoder *json.Decoder) (*agpl.Node, error) {
|
||||
var node agpl.Node
|
||||
err := decoder.Decode(&node)
|
||||
if err != nil {
|
||||
return nil, xerrors.Errorf("read json: %w", err)
|
||||
}
|
||||
|
||||
c.mutex.Lock()
|
||||
oldNode := c.nodes[id]
|
||||
if oldNode != nil {
|
||||
if oldNode.AsOf.After(node.AsOf) {
|
||||
c.mutex.Unlock()
|
||||
return oldNode, nil
|
||||
}
|
||||
}
|
||||
c.nodes[id] = &node
|
||||
connectionSockets, ok := c.agentToConnectionSockets[id]
|
||||
if !ok {
|
||||
c.mutex.Unlock()
|
||||
return &node, nil
|
||||
}
|
||||
|
||||
data, err := json.Marshal([]*agpl.Node{&node})
|
||||
if err != nil {
|
||||
c.mutex.Unlock()
|
||||
return nil, xerrors.Errorf("marshal nodes: %w", err)
|
||||
}
|
||||
|
||||
// Publish the new node to every listening socket.
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(len(connectionSockets))
|
||||
for _, connectionSocket := range connectionSockets {
|
||||
connectionSocket := connectionSocket
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
_ = connectionSocket.SetWriteDeadline(time.Now().Add(5 * time.Second))
|
||||
_, _ = connectionSocket.Write(data)
|
||||
}()
|
||||
}
|
||||
c.mutex.Unlock()
|
||||
wg.Wait()
|
||||
return &node, nil
|
||||
}
|
||||
|
||||
// Close closes all of the open connections in the coordinator and stops the
|
||||
// coordinator from accepting new connections.
|
||||
func (c *haCoordinator) Close() error {
|
||||
c.mutex.Lock()
|
||||
defer c.mutex.Unlock()
|
||||
select {
|
||||
case <-c.close:
|
||||
return nil
|
||||
default:
|
||||
}
|
||||
close(c.close)
|
||||
c.closeFunc()
|
||||
|
||||
wg := sync.WaitGroup{}
|
||||
|
||||
wg.Add(len(c.agentSockets))
|
||||
for _, socket := range c.agentSockets {
|
||||
socket := socket
|
||||
go func() {
|
||||
_ = socket.Close()
|
||||
wg.Done()
|
||||
}()
|
||||
}
|
||||
|
||||
for _, connMap := range c.agentToConnectionSockets {
|
||||
wg.Add(len(connMap))
|
||||
for _, socket := range connMap {
|
||||
socket := socket
|
||||
go func() {
|
||||
_ = socket.Close()
|
||||
wg.Done()
|
||||
}()
|
||||
}
|
||||
}
|
||||
|
||||
wg.Wait()
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *haCoordinator) publishNodesToAgent(recipient uuid.UUID, nodes []*agpl.Node) error {
|
||||
msg, err := c.formatCallMeMaybe(recipient, nodes)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("format publish message: %w", err)
|
||||
}
|
||||
|
||||
err = c.pubsub.Publish("wireguard_peers", msg)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish message: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *haCoordinator) publishAgentHello(id uuid.UUID) error {
|
||||
msg, err := c.formatAgentHello(id)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("format publish message: %w", err)
|
||||
}
|
||||
|
||||
err = c.pubsub.Publish("wireguard_peers", msg)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish message: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *haCoordinator) publishClientHello(id uuid.UUID) error {
|
||||
msg, err := c.formatClientHello(id)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("format client hello: %w", err)
|
||||
}
|
||||
err = c.pubsub.Publish("wireguard_peers", msg)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish client hello: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *haCoordinator) publishAgentToNodes(id uuid.UUID, node *agpl.Node) error {
|
||||
msg, err := c.formatAgentUpdate(id, node)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("format publish message: %w", err)
|
||||
}
|
||||
|
||||
err = c.pubsub.Publish("wireguard_peers", msg)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("publish message: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *haCoordinator) runPubsub(ctx context.Context) error {
|
||||
messageQueue := make(chan []byte, 64)
|
||||
cancelSub, err := c.pubsub.Subscribe("wireguard_peers", func(ctx context.Context, message []byte) {
|
||||
select {
|
||||
case messageQueue <- message:
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
})
|
||||
if err != nil {
|
||||
return xerrors.Errorf("subscribe wireguard peers")
|
||||
}
|
||||
go func() {
|
||||
for {
|
||||
var message []byte
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case message = <-messageQueue:
|
||||
}
|
||||
c.handlePubsubMessage(ctx, message)
|
||||
}
|
||||
}()
|
||||
|
||||
go func() {
|
||||
defer cancelSub()
|
||||
<-c.close
|
||||
}()
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *haCoordinator) handlePubsubMessage(ctx context.Context, message []byte) {
|
||||
sp := bytes.Split(message, []byte("|"))
|
||||
if len(sp) != 4 {
|
||||
c.log.Error(ctx, "invalid wireguard peer message", slog.F("msg", string(message)))
|
||||
return
|
||||
}
|
||||
|
||||
var (
|
||||
coordinatorID = sp[0]
|
||||
eventType = sp[1]
|
||||
agentID = sp[2]
|
||||
nodeJSON = sp[3]
|
||||
)
|
||||
|
||||
sender, err := uuid.ParseBytes(coordinatorID)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "invalid sender id", slog.F("id", string(coordinatorID)), slog.F("msg", string(message)))
|
||||
return
|
||||
}
|
||||
|
||||
// We sent this message!
|
||||
if sender == c.id {
|
||||
return
|
||||
}
|
||||
|
||||
switch string(eventType) {
|
||||
case "callmemaybe":
|
||||
agentUUID, err := uuid.ParseBytes(agentID)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "invalid agent id", slog.F("id", string(agentID)))
|
||||
return
|
||||
}
|
||||
|
||||
c.mutex.Lock()
|
||||
agentSocket, ok := c.agentSockets[agentUUID]
|
||||
if !ok {
|
||||
c.mutex.Unlock()
|
||||
return
|
||||
}
|
||||
c.mutex.Unlock()
|
||||
|
||||
// We get a single node over pubsub, so turn into an array.
|
||||
_, err = agentSocket.Write(nodeJSON)
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) || errors.Is(err, io.ErrClosedPipe) {
|
||||
return
|
||||
}
|
||||
c.log.Error(ctx, "send callmemaybe to agent", slog.Error(err))
|
||||
return
|
||||
}
|
||||
case "clienthello":
|
||||
agentUUID, err := uuid.ParseBytes(agentID)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "invalid agent id", slog.F("id", string(agentID)))
|
||||
return
|
||||
}
|
||||
|
||||
err = c.handleClientHello(agentUUID)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "handle agent request node", slog.Error(err))
|
||||
return
|
||||
}
|
||||
case "agenthello":
|
||||
agentUUID, err := uuid.ParseBytes(agentID)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "invalid agent id", slog.F("id", string(agentID)))
|
||||
return
|
||||
}
|
||||
|
||||
nodes := c.nodesSubscribedToAgent(agentUUID)
|
||||
if len(nodes) > 0 {
|
||||
err := c.publishNodesToAgent(agentUUID, nodes)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "publish nodes to agent", slog.Error(err))
|
||||
return
|
||||
}
|
||||
}
|
||||
case "agentupdate":
|
||||
agentUUID, err := uuid.ParseBytes(agentID)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "invalid agent id", slog.F("id", string(agentID)))
|
||||
return
|
||||
}
|
||||
|
||||
decoder := json.NewDecoder(bytes.NewReader(nodeJSON))
|
||||
_, err = c.handleAgentUpdate(agentUUID, decoder)
|
||||
if err != nil {
|
||||
c.log.Error(ctx, "handle agent update", slog.Error(err))
|
||||
return
|
||||
}
|
||||
default:
|
||||
c.log.Error(ctx, "unknown peer event", slog.F("name", string(eventType)))
|
||||
}
|
||||
}
|
||||
|
||||
// format: <coordinator id>|callmemaybe|<recipient id>|<node json>
|
||||
func (c *haCoordinator) formatCallMeMaybe(recipient uuid.UUID, nodes []*agpl.Node) ([]byte, error) {
|
||||
buf := bytes.Buffer{}
|
||||
|
||||
buf.WriteString(c.id.String() + "|")
|
||||
buf.WriteString("callmemaybe|")
|
||||
buf.WriteString(recipient.String() + "|")
|
||||
err := json.NewEncoder(&buf).Encode(nodes)
|
||||
if err != nil {
|
||||
return nil, xerrors.Errorf("encode node: %w", err)
|
||||
}
|
||||
|
||||
return buf.Bytes(), nil
|
||||
}
|
||||
|
||||
// format: <coordinator id>|agenthello|<node id>|
|
||||
func (c *haCoordinator) formatAgentHello(id uuid.UUID) ([]byte, error) {
|
||||
buf := bytes.Buffer{}
|
||||
|
||||
buf.WriteString(c.id.String() + "|")
|
||||
buf.WriteString("agenthello|")
|
||||
buf.WriteString(id.String() + "|")
|
||||
|
||||
return buf.Bytes(), nil
|
||||
}
|
||||
|
||||
// format: <coordinator id>|clienthello|<agent id>|
|
||||
func (c *haCoordinator) formatClientHello(id uuid.UUID) ([]byte, error) {
|
||||
buf := bytes.Buffer{}
|
||||
|
||||
buf.WriteString(c.id.String() + "|")
|
||||
buf.WriteString("clienthello|")
|
||||
buf.WriteString(id.String() + "|")
|
||||
|
||||
return buf.Bytes(), nil
|
||||
}
|
||||
|
||||
// format: <coordinator id>|agentupdate|<node id>|<node json>
|
||||
func (c *haCoordinator) formatAgentUpdate(id uuid.UUID, node *agpl.Node) ([]byte, error) {
|
||||
buf := bytes.Buffer{}
|
||||
|
||||
buf.WriteString(c.id.String() + "|")
|
||||
buf.WriteString("agentupdate|")
|
||||
buf.WriteString(id.String() + "|")
|
||||
err := json.NewEncoder(&buf).Encode(node)
|
||||
if err != nil {
|
||||
return nil, xerrors.Errorf("encode node: %w", err)
|
||||
}
|
||||
|
||||
return buf.Bytes(), nil
|
||||
}
|
||||
@@ -0,0 +1,261 @@
|
||||
package tailnet_test
|
||||
|
||||
import (
|
||||
"net"
|
||||
"testing"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
"cdr.dev/slog/sloggers/slogtest"
|
||||
|
||||
"github.com/coder/coder/coderd/database"
|
||||
"github.com/coder/coder/coderd/database/dbtestutil"
|
||||
"github.com/coder/coder/enterprise/tailnet"
|
||||
agpl "github.com/coder/coder/tailnet"
|
||||
"github.com/coder/coder/testutil"
|
||||
)
|
||||
|
||||
func TestCoordinatorSingle(t *testing.T) {
|
||||
t.Parallel()
|
||||
t.Run("ClientWithoutAgent", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
coordinator, err := tailnet.NewCoordinator(slogtest.Make(t, nil), database.NewPubsubInMemory())
|
||||
require.NoError(t, err)
|
||||
defer coordinator.Close()
|
||||
|
||||
client, server := net.Pipe()
|
||||
sendNode, errChan := agpl.ServeCoordinator(client, func(node []*agpl.Node) error {
|
||||
return nil
|
||||
})
|
||||
id := uuid.New()
|
||||
closeChan := make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator.ServeClient(server, id, uuid.New())
|
||||
assert.NoError(t, err)
|
||||
close(closeChan)
|
||||
}()
|
||||
sendNode(&agpl.Node{})
|
||||
require.Eventually(t, func() bool {
|
||||
return coordinator.Node(id) != nil
|
||||
}, testutil.WaitShort, testutil.IntervalFast)
|
||||
|
||||
err = client.Close()
|
||||
require.NoError(t, err)
|
||||
<-errChan
|
||||
<-closeChan
|
||||
})
|
||||
|
||||
t.Run("AgentWithoutClients", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
coordinator, err := tailnet.NewCoordinator(slogtest.Make(t, nil), database.NewPubsubInMemory())
|
||||
require.NoError(t, err)
|
||||
defer coordinator.Close()
|
||||
|
||||
client, server := net.Pipe()
|
||||
sendNode, errChan := agpl.ServeCoordinator(client, func(node []*agpl.Node) error {
|
||||
return nil
|
||||
})
|
||||
id := uuid.New()
|
||||
closeChan := make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator.ServeAgent(server, id)
|
||||
assert.NoError(t, err)
|
||||
close(closeChan)
|
||||
}()
|
||||
sendNode(&agpl.Node{})
|
||||
require.Eventually(t, func() bool {
|
||||
return coordinator.Node(id) != nil
|
||||
}, testutil.WaitShort, testutil.IntervalFast)
|
||||
err = client.Close()
|
||||
require.NoError(t, err)
|
||||
<-errChan
|
||||
<-closeChan
|
||||
})
|
||||
|
||||
t.Run("AgentWithClient", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
coordinator, err := tailnet.NewCoordinator(slogtest.Make(t, nil), database.NewPubsubInMemory())
|
||||
require.NoError(t, err)
|
||||
defer coordinator.Close()
|
||||
|
||||
agentWS, agentServerWS := net.Pipe()
|
||||
defer agentWS.Close()
|
||||
agentNodeChan := make(chan []*agpl.Node)
|
||||
sendAgentNode, agentErrChan := agpl.ServeCoordinator(agentWS, func(nodes []*agpl.Node) error {
|
||||
agentNodeChan <- nodes
|
||||
return nil
|
||||
})
|
||||
agentID := uuid.New()
|
||||
closeAgentChan := make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator.ServeAgent(agentServerWS, agentID)
|
||||
assert.NoError(t, err)
|
||||
close(closeAgentChan)
|
||||
}()
|
||||
sendAgentNode(&agpl.Node{})
|
||||
require.Eventually(t, func() bool {
|
||||
return coordinator.Node(agentID) != nil
|
||||
}, testutil.WaitShort, testutil.IntervalFast)
|
||||
|
||||
clientWS, clientServerWS := net.Pipe()
|
||||
defer clientWS.Close()
|
||||
defer clientServerWS.Close()
|
||||
clientNodeChan := make(chan []*agpl.Node)
|
||||
sendClientNode, clientErrChan := agpl.ServeCoordinator(clientWS, func(nodes []*agpl.Node) error {
|
||||
clientNodeChan <- nodes
|
||||
return nil
|
||||
})
|
||||
clientID := uuid.New()
|
||||
closeClientChan := make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator.ServeClient(clientServerWS, clientID, agentID)
|
||||
assert.NoError(t, err)
|
||||
close(closeClientChan)
|
||||
}()
|
||||
agentNodes := <-clientNodeChan
|
||||
require.Len(t, agentNodes, 1)
|
||||
sendClientNode(&agpl.Node{})
|
||||
clientNodes := <-agentNodeChan
|
||||
require.Len(t, clientNodes, 1)
|
||||
|
||||
// Ensure an update to the agent node reaches the client!
|
||||
sendAgentNode(&agpl.Node{})
|
||||
agentNodes = <-clientNodeChan
|
||||
require.Len(t, agentNodes, 1)
|
||||
|
||||
// Close the agent WebSocket so a new one can connect.
|
||||
err = agentWS.Close()
|
||||
require.NoError(t, err)
|
||||
<-agentErrChan
|
||||
<-closeAgentChan
|
||||
|
||||
// Create a new agent connection. This is to simulate a reconnect!
|
||||
agentWS, agentServerWS = net.Pipe()
|
||||
defer agentWS.Close()
|
||||
agentNodeChan = make(chan []*agpl.Node)
|
||||
_, agentErrChan = agpl.ServeCoordinator(agentWS, func(nodes []*agpl.Node) error {
|
||||
agentNodeChan <- nodes
|
||||
return nil
|
||||
})
|
||||
closeAgentChan = make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator.ServeAgent(agentServerWS, agentID)
|
||||
assert.NoError(t, err)
|
||||
close(closeAgentChan)
|
||||
}()
|
||||
// Ensure the existing listening client sends it's node immediately!
|
||||
clientNodes = <-agentNodeChan
|
||||
require.Len(t, clientNodes, 1)
|
||||
|
||||
err = agentWS.Close()
|
||||
require.NoError(t, err)
|
||||
<-agentErrChan
|
||||
<-closeAgentChan
|
||||
|
||||
err = clientWS.Close()
|
||||
require.NoError(t, err)
|
||||
<-clientErrChan
|
||||
<-closeClientChan
|
||||
})
|
||||
}
|
||||
|
||||
func TestCoordinatorHA(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
t.Run("AgentWithClient", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
_, pubsub := dbtestutil.NewDB(t)
|
||||
|
||||
coordinator1, err := tailnet.NewCoordinator(slogtest.Make(t, nil), pubsub)
|
||||
require.NoError(t, err)
|
||||
defer coordinator1.Close()
|
||||
|
||||
agentWS, agentServerWS := net.Pipe()
|
||||
defer agentWS.Close()
|
||||
agentNodeChan := make(chan []*agpl.Node)
|
||||
sendAgentNode, agentErrChan := agpl.ServeCoordinator(agentWS, func(nodes []*agpl.Node) error {
|
||||
agentNodeChan <- nodes
|
||||
return nil
|
||||
})
|
||||
agentID := uuid.New()
|
||||
closeAgentChan := make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator1.ServeAgent(agentServerWS, agentID)
|
||||
assert.NoError(t, err)
|
||||
close(closeAgentChan)
|
||||
}()
|
||||
sendAgentNode(&agpl.Node{})
|
||||
require.Eventually(t, func() bool {
|
||||
return coordinator1.Node(agentID) != nil
|
||||
}, testutil.WaitShort, testutil.IntervalFast)
|
||||
|
||||
coordinator2, err := tailnet.NewCoordinator(slogtest.Make(t, nil), pubsub)
|
||||
require.NoError(t, err)
|
||||
defer coordinator2.Close()
|
||||
|
||||
clientWS, clientServerWS := net.Pipe()
|
||||
defer clientWS.Close()
|
||||
defer clientServerWS.Close()
|
||||
clientNodeChan := make(chan []*agpl.Node)
|
||||
sendClientNode, clientErrChan := agpl.ServeCoordinator(clientWS, func(nodes []*agpl.Node) error {
|
||||
clientNodeChan <- nodes
|
||||
return nil
|
||||
})
|
||||
clientID := uuid.New()
|
||||
closeClientChan := make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator2.ServeClient(clientServerWS, clientID, agentID)
|
||||
assert.NoError(t, err)
|
||||
close(closeClientChan)
|
||||
}()
|
||||
agentNodes := <-clientNodeChan
|
||||
require.Len(t, agentNodes, 1)
|
||||
sendClientNode(&agpl.Node{})
|
||||
_ = sendClientNode
|
||||
clientNodes := <-agentNodeChan
|
||||
require.Len(t, clientNodes, 1)
|
||||
|
||||
// Ensure an update to the agent node reaches the client!
|
||||
sendAgentNode(&agpl.Node{})
|
||||
agentNodes = <-clientNodeChan
|
||||
require.Len(t, agentNodes, 1)
|
||||
|
||||
// Close the agent WebSocket so a new one can connect.
|
||||
require.NoError(t, agentWS.Close())
|
||||
require.NoError(t, agentServerWS.Close())
|
||||
<-agentErrChan
|
||||
<-closeAgentChan
|
||||
|
||||
// Create a new agent connection. This is to simulate a reconnect!
|
||||
agentWS, agentServerWS = net.Pipe()
|
||||
defer agentWS.Close()
|
||||
agentNodeChan = make(chan []*agpl.Node)
|
||||
_, agentErrChan = agpl.ServeCoordinator(agentWS, func(nodes []*agpl.Node) error {
|
||||
agentNodeChan <- nodes
|
||||
return nil
|
||||
})
|
||||
closeAgentChan = make(chan struct{})
|
||||
go func() {
|
||||
err := coordinator1.ServeAgent(agentServerWS, agentID)
|
||||
assert.NoError(t, err)
|
||||
close(closeAgentChan)
|
||||
}()
|
||||
// Ensure the existing listening client sends it's node immediately!
|
||||
clientNodes = <-agentNodeChan
|
||||
require.Len(t, clientNodes, 1)
|
||||
|
||||
err = agentWS.Close()
|
||||
require.NoError(t, err)
|
||||
<-agentErrChan
|
||||
<-closeAgentChan
|
||||
|
||||
err = clientWS.Close()
|
||||
require.NoError(t, err)
|
||||
<-clientErrChan
|
||||
<-closeClientChan
|
||||
})
|
||||
}
|
||||
Reference in New Issue
Block a user