mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
fix: Improve tailnet connections by reducing timeouts (#5043)
* fix: Improve tailnet connections by reducing timeouts This awaits connection ping before running a dial. Before, we were hitting the TCP retransmission and handshake timeouts, which could intermittently add 1 or 5 seconds to a connection being initialized. * Update Tailscale
This commit is contained in:
+57
-6
@@ -2,6 +2,7 @@ package tailnet
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
@@ -198,7 +199,7 @@ func NewConn(options *Options) (*Conn, error) {
|
||||
wireguardEngine: wireguardEngine,
|
||||
}
|
||||
wireguardEngine.SetStatusCallback(func(s *wgengine.Status, err error) {
|
||||
server.logger.Info(context.Background(), "wireguard status", slog.F("status", s), slog.F("err", err))
|
||||
server.logger.Debug(context.Background(), "wireguard status", slog.F("status", s), slog.F("err", err))
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
@@ -217,7 +218,7 @@ func NewConn(options *Options) (*Conn, error) {
|
||||
server.sendNode()
|
||||
})
|
||||
wireguardEngine.SetNetInfoCallback(func(ni *tailcfg.NetInfo) {
|
||||
server.logger.Info(context.Background(), "netinfo callback", slog.F("netinfo", ni))
|
||||
server.logger.Debug(context.Background(), "netinfo callback", slog.F("netinfo", ni))
|
||||
// If the lastMutex is blocked, it's possible that
|
||||
// multiple NetInfo callbacks occur at the same time.
|
||||
//
|
||||
@@ -383,6 +384,9 @@ func (c *Conn) UpdateNodes(nodes []*Node) error {
|
||||
if c.isClosed() {
|
||||
return nil
|
||||
}
|
||||
if errors.Is(err, wgengine.ErrNoChanges) {
|
||||
return nil
|
||||
}
|
||||
return xerrors.Errorf("reconfig: %w", err)
|
||||
}
|
||||
return nil
|
||||
@@ -395,9 +399,56 @@ func (c *Conn) Status() *ipnstate.Status {
|
||||
return sb.Status()
|
||||
}
|
||||
|
||||
// Ping sends a ping to the Wireguard engine.
|
||||
func (c *Conn) Ping(ip netip.Addr, pingType tailcfg.PingType, cb func(*ipnstate.PingResult)) {
|
||||
c.wireguardEngine.Ping(ip, pingType, cb)
|
||||
// Ping sends a Disco ping to the Wireguard engine.
|
||||
func (c *Conn) Ping(ctx context.Context, ip netip.Addr) (time.Duration, error) {
|
||||
errCh := make(chan error, 1)
|
||||
durCh := make(chan time.Duration, 1)
|
||||
go c.wireguardEngine.Ping(ip, tailcfg.PingDisco, func(pr *ipnstate.PingResult) {
|
||||
if pr.Err != "" {
|
||||
errCh <- xerrors.New(pr.Err)
|
||||
return
|
||||
}
|
||||
durCh <- time.Duration(pr.LatencySeconds * float64(time.Second))
|
||||
})
|
||||
select {
|
||||
case err := <-errCh:
|
||||
return 0, err
|
||||
case <-ctx.Done():
|
||||
return 0, ctx.Err()
|
||||
case dur := <-durCh:
|
||||
return dur, nil
|
||||
}
|
||||
}
|
||||
|
||||
// AwaitReachable pings the provided IP continually until the
|
||||
// address is reachable. It's the callers responsibility to provide
|
||||
// a timeout, otherwise this function will block forever.
|
||||
func (c *Conn) AwaitReachable(ctx context.Context, ip netip.Addr) bool {
|
||||
ticker := time.NewTicker(time.Millisecond * 100)
|
||||
defer ticker.Stop()
|
||||
completedCtx, completed := context.WithCancel(ctx)
|
||||
run := func() {
|
||||
ctx, cancelFunc := context.WithTimeout(completedCtx, time.Second)
|
||||
defer cancelFunc()
|
||||
_, err := c.Ping(ctx, ip)
|
||||
if err == nil {
|
||||
completed()
|
||||
}
|
||||
}
|
||||
go run()
|
||||
defer completed()
|
||||
for {
|
||||
select {
|
||||
case <-completedCtx.Done():
|
||||
return true
|
||||
case <-ticker.C:
|
||||
// Pings can take a while, so we can run multiple
|
||||
// in parallel to return ASAP.
|
||||
go run()
|
||||
case <-ctx.Done():
|
||||
return false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Closed is a channel that ends when the connection has
|
||||
@@ -466,7 +517,7 @@ func (c *Conn) sendNode() {
|
||||
}
|
||||
c.nodeSending = true
|
||||
go func() {
|
||||
c.logger.Info(context.Background(), "sending node", slog.F("node", node))
|
||||
c.logger.Debug(context.Background(), "sending node", slog.F("node", node))
|
||||
nodeCallback(node)
|
||||
c.lastMutex.Lock()
|
||||
c.nodeSending = false
|
||||
|
||||
@@ -62,14 +62,16 @@ func TestTailnet(t *testing.T) {
|
||||
err := w1.UpdateNodes([]*tailnet.Node{node})
|
||||
require.NoError(t, err)
|
||||
})
|
||||
|
||||
require.True(t, w2.AwaitReachable(context.Background(), w1IP))
|
||||
conn := make(chan struct{})
|
||||
go func() {
|
||||
listener, err := w1.Listen("tcp", ":35565")
|
||||
assert.NoError(t, err)
|
||||
defer listener.Close()
|
||||
nc, err := listener.Accept()
|
||||
assert.NoError(t, err)
|
||||
if !assert.NoError(t, err) {
|
||||
return
|
||||
}
|
||||
_ = nc.Close()
|
||||
conn <- struct{}{}
|
||||
}()
|
||||
|
||||
Reference in New Issue
Block a user