mirror of
https://github.com/netbirdio/netbird.git
synced 2026-10-05 21:19:08 +02:00
[client] Sweep network-bound connections when the OS switches networks
On a network switch (e.g. cellular to WiFi) the management, signal and relay sockets stay bound to the old network and look alive until the OS tears them down — measured at 5 seconds of dead air on Android, while the UI kept claiming Connected. The Android client papered over this with a full engine restart, paying for it with a torn-down TUN device and discarded peer state. Introduce client/netsweep: connections register on dial and deregister on close, and a sweep closes everything registered while aborting in-flight dials through sweep-cancellable dial contexts. The aborted dials matter: a relay dial started on the dying network would otherwise hold the reconnect loop hostage for the QUIC handshake timeout. After a sweep every failure surfaces as an ordinary read/write error and the existing retry loops redial immediately on the new network. The sweeper reaches the three long-lived connections through the same options that carry the netstate gate: a gRPC dial option wraps the management and signal transports (reconnects included), and the relay client wraps its connection in one place for the picker, the guard and foreign relays alike. Everything is nil-safe; platforms that inject no sweeper are untouched. Mobile clients expose the sweep as NotifyNetworkChange. Measured on Android against the engine restart it replaces: recovery in 1.6s instead of 3.2s, no Disconnected flash, and the TUN device, WireGuard config and peer state survive.
This commit is contained in:
@@ -14,6 +14,7 @@ import (
|
||||
|
||||
log "github.com/sirupsen/logrus"
|
||||
|
||||
"github.com/netbirdio/netbird/client/netsweep"
|
||||
auth "github.com/netbirdio/netbird/shared/relay/auth/hmac"
|
||||
"github.com/netbirdio/netbird/shared/relay/client/dialer"
|
||||
netErr "github.com/netbirdio/netbird/shared/relay/client/dialer/net"
|
||||
@@ -184,6 +185,10 @@ type Client struct {
|
||||
// datagram-sized transport is avoided on subsequent connects. Shared via
|
||||
// the manager.
|
||||
transportFallback *transportFallback
|
||||
|
||||
// sweeper cuts the relay connection on network change; the read loop
|
||||
// reports the disconnect and the guard reconnects. Shared via the manager.
|
||||
sweeper *netsweep.Sweeper
|
||||
// datagramFallbackTriggered guards a single fallback per connection so a
|
||||
// burst of oversized datagrams triggers one reconnect, not many.
|
||||
datagramFallbackTriggered atomic.Bool
|
||||
@@ -393,6 +398,11 @@ func (c *Client) Close() error {
|
||||
}
|
||||
|
||||
func (c *Client) connect(ctx context.Context) (*RelayAddr, error) {
|
||||
// A sweep cancels this context, so a dial started on the old network
|
||||
// aborts instead of waiting out its handshake timeout.
|
||||
ctx, releaseDial := c.sweeper.WrapDialContext(ctx)
|
||||
defer releaseDial()
|
||||
|
||||
mode := transportModeFromEnv()
|
||||
dialers := c.getDialers(mode)
|
||||
|
||||
@@ -417,6 +427,7 @@ func (c *Client) connect(ctx context.Context) (*RelayAddr, error) {
|
||||
return nil, fmt.Errorf("dial via FQDN: %w", err)
|
||||
}
|
||||
}
|
||||
conn = c.sweeper.WrapConn(conn)
|
||||
c.relayConn = conn
|
||||
c.datagramFallbackTriggered.Store(false)
|
||||
if tc, ok := conn.(transportConn); ok {
|
||||
|
||||
Reference in New Issue
Block a user