Files
netbird/client/internal/stdnet/stdnet.go
T
Riccardo Manfrin 3e85e40be2 [client] Cache the WireGuard interface check shared by ICE agents (#8001)
* [client] Take a WireGuard detector through the interface filter

The interface filter answers whether an interface is a WireGuard device by opening
a wgctrl client and asking for it, and it does that for every interface it is given.
Nothing about that call is tied to the caller, so it can be answered by a shared
object instead of being repeated, but the filter has no way to receive one.

InterfaceFilter and the constructors that build one now take a detector, and the ICE
config carries it so that every agent can be handed the same one. Nobody supplies a
detector yet: a nil one probes on every call, which is what the filter did before, so
this changes no behaviour.

* [client] Share one WireGuard detector across every ICE agent

Creating an ICE agent builds two interface filters, one for the agent and one for
the transport net it sits on, and each is asked about every host interface. For an
interface the disallow list does not settle, answering means opening a wgctrl client,
which builds a kernel and a userspace client and resolves the netlink family, and
then a round trip that usually just reports the device does not exist. An agent is
created per peer connection attempt, so on a large network that runs constantly:
on a routing peer with ~16000 peers it measured 2.40s of a 66.59s CPU profile, 3.6%,
split evenly between opening the client and the round trip.

The engine now owns a detector and passes it to every agent through the ICE config,
so the answer for an interface is reused instead of being asked again for each agent.
It is kept for a second, short enough that a WireGuard interface appearing is picked
up before ICE settles on candidates over it.

The callers that build one filter and keep it, the relay and the UDP mux, keep
passing nil and so keep probing, which costs them nothing at their rate.

* [client] Recheck the WireGuard cache inside the singleflight group

A caller that saw an expired entry could enter the singleflight group
after another caller had already refreshed the entry and left it, and
probe the interface a second time. Read the cache again inside the group
before probing.

This also makes the concurrent probe test independent of scheduling: a
late caller finds the fresh entry instead of starting a new probe.

* [client] Drop expired WireGuard detector entries

The detector lives as long as the engine and kept an entry for every
interface name it was ever asked about. On hosts that churn interfaces,
such as container veths, the map only grew. Remove expired entries when
a new answer is stored; the map holds a few dozen names at most, so the
sweep is cheap and runs at most once per interface per TTL.

* [client] Skip the disallow-list filter test on iOS

InterfaceFilter does not apply the disallow list on iOS, so the subtest
reaches the probe there and its no-probe assertion cannot hold.
2026-10-09 11:03:34 +02:00

253 lines
6.2 KiB
Go

// Package stdnet is an extension of the pion's stdnet.
// With it the list of the interface can come from external source.
// More info: https://github.com/golang/go/issues/40569
package stdnet
import (
"context"
"errors"
"fmt"
"net"
"net/netip"
"slices"
"strconv"
"sync"
"time"
"github.com/pion/transport/v3"
"github.com/pion/transport/v3/stdnet"
"github.com/netbirdio/netbird/client/iface/netstack"
)
const (
updateInterval = 30 * time.Second
dnsResolveTimeout = 30 * time.Second
)
var errNoSuitableAddress = errors.New("no suitable address found")
// Net is an implementation of the net.Net interface
// based on functions of the standard net package.
type Net struct {
stdnet.Net
interfaces []*transport.Interface
iFaceDiscover iFaceDiscover
// interfaceFilter should return true if the given interfaceName is allowed
interfaceFilter func(interfaceName string) bool
lastUpdate time.Time
// mu is shared between interfaces and lastUpdate
mu sync.Mutex
// ctx is the context for network operations that supports cancellation
ctx context.Context
}
// NewNetWithDiscover creates a new StdNet instance.
func NewNetWithDiscover(ctx context.Context, iFaceDiscover ExternalIFaceDiscover, disallowList []string, detector *WGDetector) *Net {
if ctx == nil {
ctx = context.Background()
}
n := &Net{
interfaceFilter: InterfaceFilter(disallowList, detector),
ctx: ctx,
}
// current ExternalIFaceDiscover implement in android-client https://github.dev/netbirdio/android-client
// so in android cli use pionDiscover
if netstack.IsEnabled() {
n.iFaceDiscover = pionDiscover{}
} else {
n.iFaceDiscover = newMobileIFaceDiscover(iFaceDiscover)
}
return n
}
// NewNet creates a new StdNet instance.
func NewNet(ctx context.Context, disallowList []string, detector *WGDetector) *Net {
if ctx == nil {
ctx = context.Background()
}
return &Net{
iFaceDiscover: pionDiscover{},
interfaceFilter: InterfaceFilter(disallowList, detector),
ctx: ctx,
}
}
// resolveAddr performs DNS resolution with context support and timeout.
func (n *Net) resolveAddr(network, address string) (netip.AddrPort, error) {
host, portStr, err := net.SplitHostPort(address)
if err != nil {
return netip.AddrPort{}, err
}
port, err := strconv.Atoi(portStr)
if err != nil {
return netip.AddrPort{}, fmt.Errorf("invalid port: %w", err)
}
if port < 0 || port > 65535 {
return netip.AddrPort{}, fmt.Errorf("invalid port: %d", port)
}
ipNet := "ip"
switch network {
case "tcp4", "udp4":
ipNet = "ip4"
case "tcp6", "udp6":
ipNet = "ip6"
}
if host == "" {
addr := netip.IPv4Unspecified()
if ipNet == "ip6" {
addr = netip.IPv6Unspecified()
}
return netip.AddrPortFrom(addr, uint16(port)), nil
}
ctx, cancel := context.WithTimeout(n.ctx, dnsResolveTimeout)
defer cancel()
addrs, err := net.DefaultResolver.LookupNetIP(ctx, ipNet, host)
if err != nil {
return netip.AddrPort{}, err
}
if len(addrs) == 0 {
return netip.AddrPort{}, errNoSuitableAddress
}
return netip.AddrPortFrom(addrs[0], uint16(port)), nil
}
// Interfaces returns a slice of interfaces which are available on the
// system
func (n *Net) Interfaces() ([]*transport.Interface, error) {
n.mu.Lock()
defer n.mu.Unlock()
iFaces, err := n.freshInterfacesLocked()
if err != nil {
return nil, err
}
return slices.Clone(iFaces), nil
}
// InterfaceByIndex returns the interface specified by index.
//
// On Solaris, it returns one of the logical network interfaces
// sharing the logical data link; for more precision use
// InterfaceByName.
func (n *Net) InterfaceByIndex(index int) (*transport.Interface, error) {
n.mu.Lock()
defer n.mu.Unlock()
iFaces, err := n.freshInterfacesLocked()
if err != nil {
return nil, err
}
for _, ifc := range iFaces {
if ifc.Index == index {
return ifc, nil
}
}
return nil, fmt.Errorf("%w: index=%d", transport.ErrInterfaceNotFound, index)
}
// InterfaceByName returns the interface specified by name.
func (n *Net) InterfaceByName(name string) (*transport.Interface, error) {
n.mu.Lock()
defer n.mu.Unlock()
iFaces, err := n.freshInterfacesLocked()
if err != nil {
return nil, err
}
for _, ifc := range iFaces {
if ifc.Name == name {
return ifc, nil
}
}
return nil, fmt.Errorf("%w: %s", transport.ErrInterfaceNotFound, name)
}
func (n *Net) freshInterfacesLocked() ([]*transport.Interface, error) {
if time.Since(n.lastUpdate) < updateInterval {
return n.interfaces, nil
}
if err := n.updateInterfacesLocked(); err != nil {
return nil, fmt.Errorf("update interfaces: %w", err)
}
return n.interfaces, nil
}
func (n *Net) updateInterfacesLocked() error {
allIFaces, err := n.iFaceDiscover.iFaces()
if err != nil {
return err
}
n.interfaces = n.filterInterfaces(allIFaces)
n.lastUpdate = time.Now()
return nil
}
func (n *Net) filterInterfaces(interfaces []*transport.Interface) []*transport.Interface {
if n.interfaceFilter == nil {
return interfaces
}
var result []*transport.Interface
for _, iface := range interfaces {
if n.interfaceFilter(iface.Name) {
result = append(result, iface)
}
}
return result
}
// ResolveUDPAddr resolves UDP addresses with context support and timeout.
func (n *Net) ResolveUDPAddr(network, address string) (*net.UDPAddr, error) {
switch network {
case "udp", "udp4", "udp6":
case "":
network = "udp"
default:
return nil, &net.OpError{Op: "resolve", Net: network, Err: net.UnknownNetworkError(network)}
}
addrPort, err := n.resolveAddr(network, address)
if err != nil {
return nil, &net.OpError{Op: "resolve", Net: network, Addr: &net.UDPAddr{IP: nil}, Err: err}
}
return net.UDPAddrFromAddrPort(addrPort), nil
}
// ResolveTCPAddr resolves TCP addresses with context support and timeout.
func (n *Net) ResolveTCPAddr(network, address string) (*net.TCPAddr, error) {
switch network {
case "tcp", "tcp4", "tcp6":
case "":
network = "tcp"
default:
return nil, &net.OpError{Op: "resolve", Net: network, Err: net.UnknownNetworkError(network)}
}
addrPort, err := n.resolveAddr(network, address)
if err != nil {
return nil, &net.OpError{Op: "resolve", Net: network, Addr: &net.TCPAddr{IP: nil}, Err: err}
}
return net.TCPAddrFromAddrPort(addrPort), nil
}