mirror of
https://github.com/netbirdio/netbird.git
synced 2026-09-12 17:59:06 +02:00
Nothing recorded which run of the connection was current, so three defects followed from the same gap. A ConnectClient is single-use, and the daemon builds a fresh one per outer-retry turn (server.go connect). Each turn overwrote s.connectClient and nothing stopped the one it replaced — the outgoing run loop had returned, which is what brought control back to the retry, but that was assumed rather than enforced, and any teardown its error path left half-done got no second chance. cleanupConnection read s.connectClient, cancelled, then stopped that engine. Nothing established the client it read was still current by the time it stopped it, so a teardown could target a client a newer turn had already replaced and leave the live one running untracked. Down's wait on clientGiveUpChan and Up's refusal to start a second loop kept the window narrow, but by arrangement rather than by construction. Third, the engine was stopped twice concurrently: actCancel woke the run loop, which stops the engine on its way out, while cleanupConnection stopped the same engine directly. The TODO there said ConnectClient.Stop was the right call and that its unbounded wait was what ruled it out. RunSupervisor records the generation of the current run. Publish refuses a client from a superseded run and stops the client it displaces, so no ConnectClient is dropped without being stopped. Stop invalidates whatever run is in flight, stops the published client and waits for the run to exit. ConnectClient.StopWithContext bounds that wait, which removes the TODO's obstacle: cleanupConnection now hands the run loop sole ownership of engine shutdown and passes Down's 5s budget down. Stop() keeps its signature and its unbounded wait, so callers outside this change are untouched. embed.Client.Stop had built the same bound by hand with a goroutine and a select purely to watch its caller's context; it passes the context down instead. clientGiveUpChan and connectClient are gone — the supervisor answers both. The MDM restart path drops its hand-rolled 10s channel wait for the same Stop, which additionally stops the client the previous run left behind. Its deliberate choice to leave clientRunning set is unchanged. Down now waits inside cleanupConnection, under s.mutex, where it previously waited after releasing it. That is what pins the client being stopped to the one current when the call started; the cost is that Down can hold the mutex for up to its 5s budget. Found while fixing the iOS wifi-to-cellular black-hole (#7329), which was the same class of defect in the mobile SDKs. No bug report backs the daemon findings — they are read off the code, and the narrow windows above may be why they have not been observed.
275 lines
8.9 KiB
Go
275 lines
8.9 KiB
Go
//go:build !android && !ios
|
|
|
|
package server
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"path/filepath"
|
|
"runtime/pprof"
|
|
"strings"
|
|
"time"
|
|
|
|
log "github.com/sirupsen/logrus"
|
|
"google.golang.org/grpc/codes"
|
|
gstatus "google.golang.org/grpc/status"
|
|
|
|
"github.com/netbirdio/netbird/client/anonymize"
|
|
"github.com/netbirdio/netbird/client/internal/debug"
|
|
"github.com/netbirdio/netbird/client/internal/ipcauth"
|
|
"github.com/netbirdio/netbird/client/proto"
|
|
mgmProto "github.com/netbirdio/netbird/shared/management/proto"
|
|
"github.com/netbirdio/netbird/version"
|
|
)
|
|
|
|
// DebugBundle creates a debug bundle and returns the location.
|
|
func (s *Server) DebugBundle(callerCtx context.Context, req *proto.DebugBundleRequest) (resp *proto.DebugBundleResponse, err error) {
|
|
if err := requirePrivilegeForUploadURL(callerCtx, req.GetUploadURL(), req.GetUploadInsecure()); err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
// The UI log is opened as whoever asked for this bundle, so a caller only
|
|
// collects a log it owns (privileged callers excepted). ok is false on a
|
|
// socket that carries no identity, which skips the UI log.
|
|
callerID, callerIdentified := ipcauth.CallerIdentity(callerCtx)
|
|
|
|
path, managementURL, err := s.generateDebugBundle(req, uiLogOpener(callerID, callerIdentified))
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
if req.GetUploadURL() == "" {
|
|
return &proto.DebugBundleResponse{Path: path}, nil
|
|
}
|
|
|
|
// The upload runs without s.mutex held: it does network I/O to a possibly
|
|
// slow destination and must not block the other RPCs that take the lock. The
|
|
// bounded context is a backstop against a hung connection.
|
|
uploadCtx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
|
defer cancel()
|
|
key, err := debug.UploadDebugBundle(uploadCtx, req.GetUploadURL(), managementURL, path, req.GetUploadInsecure())
|
|
if err != nil {
|
|
log.Errorf("failed to upload debug bundle to %s: %v", req.GetUploadURL(), err)
|
|
return &proto.DebugBundleResponse{Path: path, UploadFailureReason: err.Error()}, nil
|
|
}
|
|
|
|
log.Infof("debug bundle uploaded to %s with key %s", req.GetUploadURL(), key)
|
|
|
|
return &proto.DebugBundleResponse{Path: path, UploadedKey: key}, nil
|
|
}
|
|
|
|
// generateDebugBundle builds the bundle under s.mutex and returns its path plus
|
|
// the management URL captured under the lock, so the caller can run the upload
|
|
// without holding the lock.
|
|
func (s *Server) generateDebugBundle(req *proto.DebugBundleRequest, uiOpener debug.LogOpener) (path string, managementURL string, err error) {
|
|
s.mutex.Lock()
|
|
defer s.mutex.Unlock()
|
|
|
|
syncResponse, err := s.getLatestSyncResponse()
|
|
if err != nil {
|
|
log.Warnf("failed to get latest sync response: %v", err)
|
|
}
|
|
|
|
var clientMetrics debug.MetricsExporter
|
|
if s.runs.Current() != nil {
|
|
if engine := s.runs.Current().Engine(); engine != nil {
|
|
if cm := engine.GetClientMetrics(); cm != nil {
|
|
clientMetrics = cm
|
|
}
|
|
}
|
|
}
|
|
|
|
var cpuProfileData []byte
|
|
if s.cpuProfileBuf != nil && !s.cpuProfiling {
|
|
cpuProfileData = s.cpuProfileBuf.Bytes()
|
|
defer func() {
|
|
s.cpuProfileBuf = nil
|
|
}()
|
|
}
|
|
|
|
capturePath := s.bundleCapturePath()
|
|
defer s.cleanupBundleCapture()
|
|
|
|
var refreshStatus func()
|
|
if s.runs.Current() != nil {
|
|
engine := s.runs.Current().Engine()
|
|
if engine != nil {
|
|
refreshStatus = func() {
|
|
log.Debug("refreshing system health status for debug bundle")
|
|
// Background ctx: the bundle wants a full, fresh probe regardless
|
|
// of the DebugBundle RPC client's lifetime. The engine's own ctx
|
|
// still aborts it on shutdown.
|
|
engine.RunHealthProbes(context.Background(), true)
|
|
}
|
|
}
|
|
}
|
|
|
|
bundleGenerator := debug.NewBundleGenerator(
|
|
debug.GeneratorDependencies{
|
|
InternalConfig: s.config,
|
|
StatusRecorder: s.statusRecorder,
|
|
SyncResponse: syncResponse,
|
|
LogPath: s.logFile,
|
|
UILogPath: s.uiLogPath,
|
|
UILogOpener: uiOpener,
|
|
CPUProfile: cpuProfileData,
|
|
CapturePath: capturePath,
|
|
RefreshStatus: refreshStatus,
|
|
ClientMetrics: clientMetrics,
|
|
DaemonVersion: version.NetbirdVersion(),
|
|
CliVersion: req.CliVersion,
|
|
},
|
|
debug.BundleConfig{
|
|
Anonymize: req.GetAnonymize(),
|
|
AnonymizeLevel: anonymize.ParseLevel(req.GetAnonymizeLevel()),
|
|
IncludeSystemInfo: req.GetSystemInfo(),
|
|
LogFileCount: req.GetLogFileCount(),
|
|
},
|
|
)
|
|
|
|
path, err = bundleGenerator.Generate()
|
|
if err != nil {
|
|
return "", "", fmt.Errorf("generate debug bundle: %w", err)
|
|
}
|
|
|
|
if s.config != nil && s.config.ManagementURL != nil {
|
|
managementURL = s.config.ManagementURL.String()
|
|
}
|
|
|
|
return path, managementURL, nil
|
|
}
|
|
|
|
// GetLogLevel gets the current logging level for the server.
|
|
func (s *Server) GetLogLevel(_ context.Context, _ *proto.GetLogLevelRequest) (*proto.GetLogLevelResponse, error) {
|
|
s.mutex.Lock()
|
|
defer s.mutex.Unlock()
|
|
|
|
level := ParseLogLevel(log.GetLevel().String())
|
|
return &proto.GetLogLevelResponse{Level: level}, nil
|
|
}
|
|
|
|
// SetLogLevel sets the logging level for the server.
|
|
func (s *Server) SetLogLevel(_ context.Context, req *proto.SetLogLevelRequest) (*proto.SetLogLevelResponse, error) {
|
|
s.mutex.Lock()
|
|
defer s.mutex.Unlock()
|
|
|
|
level, err := log.ParseLevel(req.Level.String())
|
|
if err != nil {
|
|
return nil, fmt.Errorf("invalid log level: %w", err)
|
|
}
|
|
|
|
log.SetLevel(level)
|
|
|
|
if s.runs.Current() != nil {
|
|
s.runs.Current().SetLogLevel(level)
|
|
}
|
|
|
|
log.Infof("Log level set to %s", level.String())
|
|
|
|
// Signal the desktop UI so it can attach/detach its gui-client.log. Rides
|
|
// the SubscribeEvents stream as a marked event (see publishLogLevelChanged).
|
|
s.publishLogLevelChanged(level.String())
|
|
|
|
return &proto.SetLogLevelResponse{}, nil
|
|
}
|
|
|
|
// RegisterUILog records the desktop UI's absolute log path so DebugBundle can
|
|
// collect the GUI log. The daemon runs as root and can't resolve the user's
|
|
// config dir, so the UI reports it. Last-writer-wins (one UI per socket).
|
|
//
|
|
// The path arrives over an IPC any local user can reach and is later opened by
|
|
// a root daemon, so it is constrained to the file name the UI writes and to a
|
|
// local absolute path. Authorization happens when DebugBundle opens it: the
|
|
// bundle refuses a file its requester does not own. A caller the daemon cannot
|
|
// identify cannot register a path at all.
|
|
func (s *Server) RegisterUILog(callerCtx context.Context, req *proto.RegisterUILogRequest) (*proto.RegisterUILogResponse, error) {
|
|
if _, ok := ipcauth.CallerIdentity(callerCtx); !ok {
|
|
return nil, gstatus.Error(codes.PermissionDenied,
|
|
"registering a UI log path requires a control channel that carries the caller's identity")
|
|
}
|
|
|
|
path := filepath.Clean(req.GetPath())
|
|
if !filepath.IsAbs(path) || filepath.Base(path) != uiLogFileName {
|
|
return nil, gstatus.Errorf(codes.InvalidArgument, "UI log path must be an absolute path ending in %s", uiLogFileName)
|
|
}
|
|
// filepath.IsAbs accepts a Windows UNC path (\\host\share\...) and a device
|
|
// path (\\.\, \\?\); opening one would make the root daemon reach a remote
|
|
// or device namespace. Require a plain local path.
|
|
if strings.HasPrefix(path, `\\`) {
|
|
return nil, gstatus.Error(codes.InvalidArgument, "UI log path must be a local path, not a UNC or device path")
|
|
}
|
|
|
|
s.mutex.Lock()
|
|
defer s.mutex.Unlock()
|
|
|
|
s.uiLogPath = path
|
|
log.Infof("registered UI log path %s", s.uiLogPath)
|
|
|
|
return &proto.RegisterUILogResponse{}, nil
|
|
}
|
|
|
|
// SetSyncResponsePersistence sets the sync response persistence for the server.
|
|
func (s *Server) SetSyncResponsePersistence(_ context.Context, req *proto.SetSyncResponsePersistenceRequest) (*proto.SetSyncResponsePersistenceResponse, error) {
|
|
s.mutex.Lock()
|
|
defer s.mutex.Unlock()
|
|
|
|
enabled := req.GetEnabled()
|
|
s.persistSyncResponse = enabled
|
|
if s.runs.Current() != nil {
|
|
s.runs.Current().SetSyncResponsePersistence(enabled)
|
|
}
|
|
|
|
return &proto.SetSyncResponsePersistenceResponse{}, nil
|
|
}
|
|
|
|
func (s *Server) getLatestSyncResponse() (*mgmProto.SyncResponse, error) {
|
|
cClient := s.runs.Current()
|
|
if cClient == nil {
|
|
return nil, errors.New("connect client is not initialized")
|
|
}
|
|
|
|
return cClient.GetLatestSyncResponse()
|
|
}
|
|
|
|
// StartCPUProfile starts CPU profiling in the daemon.
|
|
func (s *Server) StartCPUProfile(_ context.Context, _ *proto.StartCPUProfileRequest) (*proto.StartCPUProfileResponse, error) {
|
|
s.mutex.Lock()
|
|
defer s.mutex.Unlock()
|
|
|
|
if s.cpuProfiling {
|
|
return nil, fmt.Errorf("CPU profiling already in progress")
|
|
}
|
|
|
|
s.cpuProfileBuf = &bytes.Buffer{}
|
|
s.cpuProfiling = true
|
|
if err := pprof.StartCPUProfile(s.cpuProfileBuf); err != nil {
|
|
s.cpuProfileBuf = nil
|
|
s.cpuProfiling = false
|
|
return nil, fmt.Errorf("start CPU profile: %w", err)
|
|
}
|
|
|
|
log.Info("CPU profiling started")
|
|
return &proto.StartCPUProfileResponse{}, nil
|
|
}
|
|
|
|
// StopCPUProfile stops CPU profiling in the daemon.
|
|
func (s *Server) StopCPUProfile(_ context.Context, _ *proto.StopCPUProfileRequest) (*proto.StopCPUProfileResponse, error) {
|
|
s.mutex.Lock()
|
|
defer s.mutex.Unlock()
|
|
|
|
if !s.cpuProfiling {
|
|
return nil, fmt.Errorf("CPU profiling not in progress")
|
|
}
|
|
|
|
pprof.StopCPUProfile()
|
|
s.cpuProfiling = false
|
|
|
|
if s.cpuProfileBuf != nil {
|
|
log.Infof("CPU profiling stopped, captured %d bytes", s.cpuProfileBuf.Len())
|
|
}
|
|
|
|
return &proto.StopCPUProfileResponse{}, nil
|
|
}
|