mirror of
https://github.com/netbirdio/netbird.git
synced 2026-09-15 11:19:08 +02:00
The agent network gateway service is synthesised as private: agents reach it over the WireGuard tunnel, authorised by their peer identity, with the cluster itself as its only target. Only a reverse proxy cluster with private capabilities can serve that, which management reports per cluster as the `private` capability — the same supports_private flag the dashboard gates NetBird-only services on. CreateSettings took any hostname as proxy_address and only normalised it, so a bootstrap could pin an account to a cluster without private capabilities. The endpoint is immutable, leaving a dead gateway until the settings row is deleted and re-bootstrapped. Both bootstrap paths now validate the picked cluster before an endpoint is allocated: a cluster the account can see must have a connected proxy reporting the capability, shared and account-owned alike. Whether management knows a cluster comes from its proxy rows, never from how fresh their heartbeats are, so a cluster without the capability stays refused while its proxies are merely offline. A hostname no proxy has declared stays pinnable, the address-first order the self-addressed path documents. Cluster identity is compared case-insensitively over the account's cluster list, since proxies declare their address as the operator spelled it. Rebased onto main after #7519 landed; the ownership check this builds on is main's now. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Sa3DsBDP3VciAi4PPG17L6
179 lines
6.9 KiB
Go
179 lines
6.9 KiB
Go
//go:build e2e
|
|
|
|
package agentnetwork
|
|
|
|
import (
|
|
"context"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/stretchr/testify/assert"
|
|
"github.com/stretchr/testify/require"
|
|
|
|
"github.com/netbirdio/netbird/e2e/harness"
|
|
"github.com/netbirdio/netbird/shared/management/http/api"
|
|
)
|
|
|
|
// TestSettingsBootstrapValidatesProxyCluster covers the bootstrap-time check
|
|
// on the picked cluster, end to end against a real proxy.
|
|
//
|
|
// The synthesised gateway service is always private: agents reach it over the
|
|
// WireGuard tunnel and are authorised by their peer identity. Only a cluster
|
|
// with private capabilities can serve that, and management reports it per
|
|
// cluster as the `private` capability — the same supports_private flag the
|
|
// dashboard reads to decide which clusters it may offer. The endpoint assigned at bootstrap is immutable, so pinning
|
|
// to a cluster that cannot serve it has to be refused up front rather than
|
|
// leaving the account with a dead gateway.
|
|
//
|
|
// One combined server and one cluster address, walked through three states:
|
|
// a live centralised proxy (refused), that proxy stopped so nothing in the
|
|
// cluster is live any more (still refused — the record of what the cluster is
|
|
// outlives its heartbeats), and finally a private-capable proxy (accepted, the
|
|
// capability being any-true across the cluster's live proxies). Same account,
|
|
// same address, so nothing but the cluster's state accounts for the different
|
|
// answers.
|
|
func TestSettingsBootstrapValidatesProxyCluster(t *testing.T) {
|
|
ctx := context.Background()
|
|
|
|
fresh, err := harnessStartFresh(ctx, t)
|
|
require.NoError(t, err, "start dedicated combined server")
|
|
|
|
proxyToken, err := fresh.CreateProxyTokenCLI(ctx, "e2e-cluster-validation")
|
|
require.NoError(t, err, "mint proxy token via CLI")
|
|
|
|
const cluster = harness.AgentNetworkCluster
|
|
|
|
// A centralised proxy: connected and serving the cluster, but without
|
|
// private capabilities, so it cannot serve a private service.
|
|
central, err := harness.StartProxy(ctx, fresh, proxyToken, map[string]string{
|
|
"NB_PROXY_PRIVATE": "false",
|
|
})
|
|
require.NoError(t, err, "start centralised proxy")
|
|
// Terminated mid-test; the cleanup only covers an early failure.
|
|
t.Cleanup(func() { _ = central.Terminate(context.Background()) })
|
|
|
|
waitClusterPrivate(ctx, t, fresh, cluster, false)
|
|
|
|
_, err = fresh.CreateSettings(ctx, api.AgentNetworkSettingsCreateRequest{
|
|
ProxyAddress: ptr(cluster),
|
|
})
|
|
require.Error(t, err, "bootstrap onto a cluster without private capabilities must be refused")
|
|
requireClientError(t, err)
|
|
assert.Contains(t, err.Error(), "private capabilities",
|
|
"the refusal must name what the cluster is missing: %v", err)
|
|
|
|
after, err := fresh.GetSettings(ctx)
|
|
require.NoError(t, err, "settings must still read after a refused bootstrap")
|
|
assert.Empty(t, after.Endpoint, "a refused bootstrap must not assign an endpoint")
|
|
assert.Empty(t, after.ProxyAddress, "a refused bootstrap must not pin a cluster")
|
|
|
|
// Stopping the centralised proxy must not turn the refusal into an
|
|
// acceptance: the cluster's proxy rows outlive their heartbeats (only the
|
|
// hourly stale reaper removes them), so the cluster is still on record as
|
|
// one that cannot serve the gateway. Judging on liveness instead would
|
|
// make "wait for the proxy to go quiet" a way to pin the account's
|
|
// immutable endpoint to a cluster that can never serve it.
|
|
require.NoError(t, central.Terminate(ctx), "stop the centralised proxy")
|
|
waitClusterAbsent(ctx, t, fresh, cluster)
|
|
|
|
_, err = fresh.CreateSettings(ctx, api.AgentNetworkSettingsCreateRequest{
|
|
ProxyAddress: ptr(cluster),
|
|
})
|
|
require.Error(t, err, "an offline cluster without private capabilities on record must stay refused")
|
|
requireClientError(t, err)
|
|
|
|
// Add a private-capable proxy to the same cluster: now it can serve a private
|
|
// service, and the very same request must go through.
|
|
privateProxy, err := harness.StartProxy(ctx, fresh, proxyToken)
|
|
require.NoError(t, err, "start private-capable proxy")
|
|
t.Cleanup(func() { _ = privateProxy.Terminate(context.Background()) })
|
|
|
|
waitClusterPrivate(ctx, t, fresh, cluster, true)
|
|
|
|
bootstrapped, err := fresh.CreateSettings(ctx, api.AgentNetworkSettingsCreateRequest{
|
|
ProxyAddress: ptr(cluster),
|
|
})
|
|
require.NoError(t, err, "bootstrap onto a private-capable cluster must succeed")
|
|
assert.Equal(t, cluster, bootstrapped.ProxyAddress, "the pinned cluster is the requested one")
|
|
assert.True(t, strings.HasSuffix(bootstrapped.Endpoint, "."+cluster),
|
|
"the endpoint must hang one label beneath the cluster: %s", bootstrapped.Endpoint)
|
|
}
|
|
|
|
// waitClusterPrivate polls the domains endpoint — the list the dashboard picks
|
|
// its bootstrap cluster from — until the free domain for clusterAddr reports
|
|
// supports_private == want. A proxy's capabilities land when it registers, so
|
|
// this is the barrier between starting a proxy and asserting on what
|
|
// management thinks its cluster can do.
|
|
func waitClusterPrivate(ctx context.Context, t *testing.T, c *harness.Combined, clusterAddr string, want bool) {
|
|
t.Helper()
|
|
|
|
deadline := time.Now().Add(90 * time.Second)
|
|
var last string
|
|
for time.Now().Before(deadline) {
|
|
domains, err := c.API().ReverseProxyDomains.List(ctx)
|
|
if err != nil {
|
|
last = "list domains: " + err.Error()
|
|
} else {
|
|
last = "cluster not listed"
|
|
for _, d := range domains {
|
|
if d.Domain != clusterAddr {
|
|
continue
|
|
}
|
|
if d.SupportsPrivate == nil {
|
|
last = "supports_private not reported yet"
|
|
break
|
|
}
|
|
if *d.SupportsPrivate == want {
|
|
return
|
|
}
|
|
last = "supports_private is not the expected value"
|
|
break
|
|
}
|
|
}
|
|
if !waitBeforeRetry(ctx, 2*time.Second) {
|
|
break
|
|
}
|
|
}
|
|
t.Fatalf("cluster %s never reported supports_private=%v: %s", clusterAddr, want, last)
|
|
}
|
|
|
|
// waitClusterAbsent polls the domains endpoint until clusterAddr is no longer
|
|
// offered, i.e. management sees no live proxy in it. The free-domain list is
|
|
// built from the active clusters, so this is how a proxy going away becomes
|
|
// observable — while the cluster's rows, and so its capability record, remain.
|
|
//
|
|
// The budget has to clear the active window, not just the disconnect: a proxy
|
|
// that closes its stream cleanly is marked disconnected at once, but one that
|
|
// dies without that is only dropped when its last heartbeat ages past
|
|
// proxyActiveThreshold (2 minutes), so a 90s deadline could fail the test on
|
|
// the slow path alone.
|
|
func waitClusterAbsent(ctx context.Context, t *testing.T, c *harness.Combined, clusterAddr string) {
|
|
t.Helper()
|
|
|
|
deadline := time.Now().Add(3 * time.Minute)
|
|
var last string
|
|
for time.Now().Before(deadline) {
|
|
domains, err := c.API().ReverseProxyDomains.List(ctx)
|
|
if err != nil {
|
|
last = "list domains: " + err.Error()
|
|
} else {
|
|
listed := false
|
|
for _, d := range domains {
|
|
if d.Domain == clusterAddr {
|
|
listed = true
|
|
break
|
|
}
|
|
}
|
|
if !listed {
|
|
return
|
|
}
|
|
last = "cluster still listed as active"
|
|
}
|
|
if !waitBeforeRetry(ctx, 2*time.Second) {
|
|
break
|
|
}
|
|
}
|
|
t.Fatalf("cluster %s never dropped out of the active list: %s", clusterAddr, last)
|
|
}
|