Files
netbird/client/internal/engine_certposture.go
T

160 lines
5.8 KiB
Go

package internal
import (
"context"
"errors"
"fmt"
"sync"
"time"
log "github.com/sirupsen/logrus"
"github.com/netbirdio/netbird/client/internal/certproof"
cProto "github.com/netbirdio/netbird/client/proto"
"github.com/netbirdio/netbird/client/system"
mgmProto "github.com/netbirdio/netbird/shared/management/proto"
)
const (
// certContextPollInterval is how often the engine checks whether the user who can
// answer certificate challenges has changed, such as a login after an autostart.
certContextPollInterval = time.Minute
// certRetryInterval is how long after a collection that proved nothing it is tried
// again with the same user, for stores that come up late: a keychain unlocked after
// login, or a TPM resource manager started after the daemon.
certRetryInterval = 5 * time.Minute
)
// errSystemInfoTimeout reports that gathering the system info for a meta sync timed out,
// so the sync was skipped rather than holding syncMsgMux on a stuck system call.
var errSystemInfoTimeout = errors.New("system info gathering timed out")
// certPostureState remembers what the last certificate proof collection saw, so the
// engine can collect again when it went stale and tell the user when proofs go missing.
type certPostureState struct {
mu sync.Mutex
attempted bool
attemptedAt time.Time
userContext string
proven bool
}
// record stores the outcome of a collection and reports whether it changed from proven
// to unproven or back. The first collection only counts as a change when it proved nothing.
func (s *certPostureState) record(userContext string, proven bool, now time.Time) (changed bool) {
s.mu.Lock()
defer s.mu.Unlock()
changed = (s.attempted && s.proven != proven) || (!s.attempted && !proven)
s.attempted = true
s.attemptedAt = now
s.userContext = userContext
s.proven = proven
return changed
}
// stale reports whether the last collection no longer reflects what the device can
// prove: the user who can answer has changed, or nothing was proven and the retry
// interval has passed. Before the first collection the sync itself collects.
func (s *certPostureState) stale(userContext string, now time.Time) bool {
s.mu.Lock()
defer s.mu.Unlock()
if !s.attempted {
return false
}
if userContext != s.userContext {
return true
}
return !s.proven && now.Sub(s.attemptedAt) >= certRetryInterval
}
// attachCertificateProofs answers the certificate challenges in checks with the
// certificates reachable on this device, signing each challenge nonce for our peer key.
// Collection is bounded in time because callers hold the sync loop while it runs.
func (e *Engine) attachCertificateProofs(info *system.Info, checks []*mgmProto.Checks) {
if !certproof.HasChallenges(checks) {
info.CertificateProofs = nil
return
}
userContext := certproof.UserContext(e.config.CertStore)
peerKey := e.config.WgPrivateKey.PublicKey()
info.CertificateProofs = e.certProofs.Collect(e.ctx, checks, peerKey[:], e.config.CertStore)
proven := len(info.CertificateProofs) > 0
if e.certState.record(userContext, proven, time.Now()) {
e.publishCertificatePostureEvent(proven)
}
}
// publishCertificatePostureEvent tells the user when the device stops proving any
// certificate, which management treats as failing every certificate posture check, and
// when it proves one again. Without it the loss of access would have no visible cause.
func (e *Engine) publishCertificatePostureEvent(proven bool) {
if e.statusRecorder == nil {
return
}
if proven {
e.statusRecorder.PublishEvent(cProto.SystemEvent_INFO, cProto.SystemEvent_SYSTEM,
"certificate posture: a certificate is proven again",
"A certificate required by your organization's device policy is available again.", nil)
return
}
e.statusRecorder.PublishEvent(cProto.SystemEvent_WARNING, cProto.SystemEvent_SYSTEM,
"certificate posture: no certificate could be proven",
"NetBird could not use a certificate required by your organization's device policy. "+
"Access to some resources may be blocked until one is available.", nil)
}
// watchCertificatePosture collects certificate proofs again when the last collection
// went stale, until ctx is done. Proofs are otherwise only collected when the checks
// change or the sync stream reconnects, so a daemon started before anyone logged in
// would not prove a user certificate until the next network map.
func (e *Engine) watchCertificatePosture(ctx context.Context) {
ticker := time.NewTicker(certContextPollInterval)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
return
case <-ticker.C:
if err := e.recollectCertificateProofsIfStale(); err != nil && !errors.Is(err, errSystemInfoTimeout) {
log.Warnf("failed to refresh certificate posture proofs: %v", err)
}
}
}
}
func (e *Engine) recollectCertificateProofsIfStale() error {
userContext := certproof.UserContext(e.config.CertStore)
if !e.certState.stale(userContext, time.Now()) {
return nil
}
e.syncMsgMux.Lock()
defer e.syncMsgMux.Unlock()
if e.ctx.Err() != nil || !certproof.HasChallenges(e.checks) {
return nil
}
log.Debugf("certificate posture: proofs are stale, collecting again")
return e.syncChecksMeta(e.checks)
}
// syncChecksMeta gathers the system info that checks evaluate, with its certificate
// proofs, and sends it to management. The caller holds syncMsgMux.
func (e *Engine) syncChecksMeta(checks []*mgmProto.Checks) error {
info, ok := e.infoSource.Refresh(e.ctx, systemInfoTimeout, checks, e.overlayAddresses()...)
if !ok {
// Gathering timed out; skip the meta sync this cycle rather than blocking the
// sync loop (and syncMsgMux) on a stuck system call. A later sync will retry.
return errSystemInfoTimeout
}
e.applyInfoFlags(info)
e.attachCertificateProofs(info, checks)
if err := e.mgmClient.SyncMeta(info); err != nil {
return fmt.Errorf("sync meta: %w", err)
}
return nil
}