Compare commits

..

3 Commits

Author SHA1 Message Date
John Maguire 8589b76e1f Do not warn about local address failures on every update
localAddrs runs inside every SendUpdate, which fires on lighthouse.interval
and again on every network change via RebindUDPServer. Logging the failure
there meant a persistent failure warned every interval for the life of the
process, once per failing interface.

collectLocalAddrs now returns its failures instead of logging them, which
keeps it stateless and lets the tests assert on errors rather than log
output. The lighthouse holds the previous error and warns only when it
changes, demoting repeats to Debug, matching how handshake_manager handles
repeated send failures.
2026-07-24 17:53:55 -04:00
John Maguire 28cff022ee Link an android binary in build-test-mobile
The package builds only compile, so a linker-only failure could not fail
CI. anet relies on //go:linkname, which the linker rejects on Go 1.23+
without -checklinkname=0, so linking cmd/nebula for android both covers
that regression and records the flag requirement.
2026-07-24 17:53:55 -04:00
John Maguire 8cbee0e965 Advertise underlay addresses on Android
On Android 11+ the app sandbox denies bind() on netlink_route_socket, so
the stdlib's net.Interfaces fails with EACCES. localAddrs discarded that
error and returned an empty slice, so the node advertised no underlay
addresses and peers could only ever reach it at the address a lighthouse
observed. A device on the same LAN as a peer was unreachable at its LAN
address.

Split interface enumeration behind a build-tagged seam and use
github.com/wlynxg/anet on Android, which reads RTM_GETADDR from an
unbound socket. Interface addresses have to come from anet as well, since
net.Interface.Addrs goes back through the same denied path. Every other
platform keeps the net package implementation.

Stop discarding the enumeration errors, which are exceptional now that
the sandbox case is handled.

anet needs -ldflags=-checklinkname=0 on Go 1.23+. Nebula ships no Android
binaries, so build-test-mobile is unaffected, but consumers linking
Android artifacts will need the flag.
2026-07-24 15:06:29 -04:00
16 changed files with 282 additions and 234 deletions
+3
View File
@@ -222,11 +222,14 @@ test-cov-html:
go test -coverprofile=coverage.out go test -coverprofile=coverage.out
go tool cover -html=coverage.out go tool cover -html=coverage.out
# The package builds only compile. The final line links an android binary so a linker-only failure,
# such as the //go:linkname reference anet makes, cannot pass CI.
build-test-mobile: build-test-mobile:
GOARCH=amd64 GOOS=ios go build $(shell go list ./... | grep -v '/cmd/\|/examples/') GOARCH=amd64 GOOS=ios go build $(shell go list ./... | grep -v '/cmd/\|/examples/')
GOARCH=arm64 GOOS=ios go build $(shell go list ./... | grep -v '/cmd/\|/examples/') GOARCH=arm64 GOOS=ios go build $(shell go list ./... | grep -v '/cmd/\|/examples/')
GOARCH=amd64 GOOS=android go build $(shell go list ./... | grep -v '/cmd/\|/examples/') GOARCH=amd64 GOOS=android go build $(shell go list ./... | grep -v '/cmd/\|/examples/')
GOARCH=arm64 GOOS=android go build $(shell go list ./... | grep -v '/cmd/\|/examples/') GOARCH=arm64 GOOS=android go build $(shell go list ./... | grep -v '/cmd/\|/examples/')
GOARCH=arm64 GOOS=android go build -ldflags=-checklinkname=0 -o /dev/null ${NEBULA_CMD_PATH}
bench: bench:
go test -bench=. go test -bench=.
+8 -13
View File
@@ -105,17 +105,11 @@ func (cm *connectionManager) getInactivityTimeout() time.Duration {
} }
func (cm *connectionManager) In(h *HostInfo) { func (cm *connectionManager) In(h *HostInfo) {
h.markIn() h.in.Store(true)
} }
// OutRelay records relayed traffic, leaving the rebind epoch for the direct path to this host to consume func (cm *connectionManager) Out(h *HostInfo) {
func (cm *connectionManager) OutRelay(h *HostInfo) { h.out.Store(true)
h.markOutOnly()
}
// Out records outbound traffic and reports whether we rebound since this tunnel last sent
func (cm *connectionManager) Out(h *HostInfo) bool {
return h.markOut(cm.intf.rebindEpoch.Load())
} }
func (cm *connectionManager) RelayUsed(localIndex uint32) { func (cm *connectionManager) RelayUsed(localIndex uint32) {
@@ -134,7 +128,8 @@ func (cm *connectionManager) RelayUsed(localIndex uint32) {
// getAndResetTrafficCheck returns if there was any inbound or outbound traffic within the last tick and // getAndResetTrafficCheck returns if there was any inbound or outbound traffic within the last tick and
// resets the state for this local index // resets the state for this local index
func (cm *connectionManager) getAndResetTrafficCheck(h *HostInfo, now time.Time) (bool, bool) { func (cm *connectionManager) getAndResetTrafficCheck(h *HostInfo, now time.Time) (bool, bool) {
in, out := h.takeTraffic() in := h.in.Swap(false)
out := h.out.Swap(false)
if in || out { if in || out {
h.lastUsed = now h.lastUsed = now
} }
@@ -345,7 +340,7 @@ func (cm *connectionManager) makeTrafficDecision(localIndex uint32, now time.Tim
"tunnelCheck", m{"state": "alive", "method": "passive"}, "tunnelCheck", m{"state": "alive", "method": "passive"},
) )
} }
hostinfo.setPendingDeletion(false) hostinfo.pendingDeletion.Store(false)
if mainHostInfo { if mainHostInfo {
decision = tryRehandshake decision = tryRehandshake
@@ -368,7 +363,7 @@ func (cm *connectionManager) makeTrafficDecision(localIndex uint32, now time.Tim
return decision, hostinfo, primary return decision, hostinfo, primary
} }
if hostinfo.isPendingDeletion() { if hostinfo.pendingDeletion.Load() {
// We have already sent a test packet and nothing was returned, this hostinfo is dead // We have already sent a test packet and nothing was returned, this hostinfo is dead
hostinfo.logger(cm.l).Info("Tunnel status", hostinfo.logger(cm.l).Info("Tunnel status",
"tunnelCheck", m{"state": "dead", "method": "active"}, "tunnelCheck", m{"state": "dead", "method": "active"},
@@ -419,7 +414,7 @@ func (cm *connectionManager) makeTrafficDecision(localIndex uint32, now time.Tim
} }
} }
hostinfo.setPendingDeletion(true) hostinfo.pendingDeletion.Store(true)
cm.trafficTimer.Add(hostinfo.localIndexId, cm.pendingDeletionInterval) cm.trafficTimer.Add(hostinfo.localIndexId, cm.pendingDeletionInterval)
return decision, hostinfo, nil return decision, hostinfo, nil
} }
+36 -36
View File
@@ -86,25 +86,25 @@ func Test_NewConnectionManagerTest(t *testing.T) {
// We saw traffic out to vpnIp // We saw traffic out to vpnIp
nc.Out(hostinfo) nc.Out(hostinfo)
nc.In(hostinfo) nc.In(hostinfo)
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0]) assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId) assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId)
assert.True(t, hostinfo.sentSinceCheck()) assert.True(t, hostinfo.out.Load())
assert.True(t, (hostinfo.state.Load()&stateIn != 0)) assert.True(t, hostinfo.in.Load())
// Do a traffic check tick, should not be pending deletion but should not have any in/out packets recorded // Do a traffic check tick, should not be pending deletion but should not have any in/out packets recorded
nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now()) nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now())
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
// Do another traffic check tick, this host should be pending deletion now // Do another traffic check tick, this host should be pending deletion now
nc.Out(hostinfo) nc.Out(hostinfo)
assert.True(t, hostinfo.sentSinceCheck()) assert.True(t, hostinfo.out.Load())
nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now()) nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now())
assert.True(t, hostinfo.isPendingDeletion()) assert.True(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId) assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId)
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0]) assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
@@ -168,33 +168,33 @@ func Test_NewConnectionManagerTest2(t *testing.T) {
// We saw traffic out to vpnIp // We saw traffic out to vpnIp
nc.Out(hostinfo) nc.Out(hostinfo)
nc.In(hostinfo) nc.In(hostinfo)
assert.True(t, (hostinfo.state.Load()&stateIn != 0)) assert.True(t, hostinfo.in.Load())
assert.True(t, hostinfo.sentSinceCheck()) assert.True(t, hostinfo.out.Load())
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0]) assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId) assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId)
// Do a traffic check tick, should not be pending deletion but should not have any in/out packets recorded // Do a traffic check tick, should not be pending deletion but should not have any in/out packets recorded
nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now()) nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now())
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
// Do another traffic check tick, this host should be pending deletion now // Do another traffic check tick, this host should be pending deletion now
nc.Out(hostinfo) nc.Out(hostinfo)
nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now()) nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now())
assert.True(t, hostinfo.isPendingDeletion()) assert.True(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId) assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId)
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0]) assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
// We saw traffic, should no longer be pending deletion // We saw traffic, should no longer be pending deletion
nc.In(hostinfo) nc.In(hostinfo)
nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now()) nc.doTrafficCheck(hostinfo.localIndexId, p, nb, out, time.Now())
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId) assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId)
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0]) assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
} }
@@ -253,31 +253,31 @@ func Test_NewConnectionManager_DisconnectInactive(t *testing.T) {
// Do a traffic check tick, in and out should be cleared but should not be pending deletion // Do a traffic check tick, in and out should be cleared but should not be pending deletion
nc.Out(hostinfo) nc.Out(hostinfo)
nc.In(hostinfo) nc.In(hostinfo)
assert.True(t, hostinfo.sentSinceCheck()) assert.True(t, hostinfo.out.Load())
assert.True(t, (hostinfo.state.Load()&stateIn != 0)) assert.True(t, hostinfo.in.Load())
now := time.Now() now := time.Now()
decision, _, _ := nc.makeTrafficDecision(hostinfo.localIndexId, now) decision, _, _ := nc.makeTrafficDecision(hostinfo.localIndexId, now)
assert.Equal(t, tryRehandshake, decision) assert.Equal(t, tryRehandshake, decision)
assert.Equal(t, now, hostinfo.lastUsed) assert.Equal(t, now, hostinfo.lastUsed)
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
decision, _, _ = nc.makeTrafficDecision(hostinfo.localIndexId, now.Add(time.Second*5)) decision, _, _ = nc.makeTrafficDecision(hostinfo.localIndexId, now.Add(time.Second*5))
assert.Equal(t, doNothing, decision) assert.Equal(t, doNothing, decision)
assert.Equal(t, now, hostinfo.lastUsed) assert.Equal(t, now, hostinfo.lastUsed)
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
// Do another traffic check tick, should still not be pending deletion // Do another traffic check tick, should still not be pending deletion
decision, _, _ = nc.makeTrafficDecision(hostinfo.localIndexId, now.Add(time.Second*10)) decision, _, _ = nc.makeTrafficDecision(hostinfo.localIndexId, now.Add(time.Second*10))
assert.Equal(t, doNothing, decision) assert.Equal(t, doNothing, decision)
assert.Equal(t, now, hostinfo.lastUsed) assert.Equal(t, now, hostinfo.lastUsed)
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId) assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId)
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0]) assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
@@ -285,9 +285,9 @@ func Test_NewConnectionManager_DisconnectInactive(t *testing.T) {
decision, _, _ = nc.makeTrafficDecision(hostinfo.localIndexId, now.Add(time.Minute*10)) decision, _, _ = nc.makeTrafficDecision(hostinfo.localIndexId, now.Add(time.Minute*10))
assert.Equal(t, closeTunnel, decision) assert.Equal(t, closeTunnel, decision)
assert.Equal(t, now, hostinfo.lastUsed) assert.Equal(t, now, hostinfo.lastUsed)
assert.False(t, hostinfo.isPendingDeletion()) assert.False(t, hostinfo.pendingDeletion.Load())
assert.False(t, hostinfo.sentSinceCheck()) assert.False(t, hostinfo.out.Load())
assert.False(t, (hostinfo.state.Load()&stateIn != 0)) assert.False(t, hostinfo.in.Load())
assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId) assert.Contains(t, nc.hostMap.Indexes, hostinfo.localIndexId)
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0]) assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
} }
+1 -1
View File
@@ -212,7 +212,7 @@ func (c *Control) RebindUDPServer() {
c.f.lightHouse.SendUpdate() c.f.lightHouse.SendUpdate()
// Let the main interface know that we rebound so that underlying tunnels know to trigger punches from their remotes // Let the main interface know that we rebound so that underlying tunnels know to trigger punches from their remotes
c.f.rebindEpoch.Add(1) c.f.rebindCount++
} }
// ListHostmapHosts returns details about the actual or pending (handshaking) hostmap by vpn ip // ListHostmapHosts returns details about the actual or pending (handshaking) hostmap by vpn ip
-10
View File
@@ -123,16 +123,6 @@ func (c *Control) SetLocalAddrsFn(fn func(*LocalAllowList) []netip.Addr) {
c.f.lightHouse.localAddrsFn = fn c.f.lightHouse.localAddrsFn = fn
} }
// GetRebindEpochFor returns the rebind epoch a tunnel last sent under, so a test can tell whether a send
// consumed the epoch edge without having to infer it from lighthouse traffic.
func (c *Control) GetRebindEpochFor(vpnAddr netip.Addr) (uint32, bool) {
h := c.f.hostMap.QueryVpnAddr(vpnAddr)
if h == nil {
return 0, false
}
return h.state.Load() >> stateEpochShift, true
}
func (c *Control) KillPendingTunnel(vpnIp netip.Addr) bool { func (c *Control) KillPendingTunnel(vpnIp netip.Addr) bool {
hostinfo := c.f.handshakeManager.QueryVpnAddr(vpnIp) hostinfo := c.f.handshakeManager.QueryVpnAddr(vpnIp)
if hostinfo == nil { if hostinfo == nil {
-54
View File
@@ -223,57 +223,3 @@ func TestRebindAdvertisesNewAddressAfterMove(t *testing.T) {
lhControl.Stop() lhControl.Stop()
myControl.Stop() myControl.Stop()
} }
// A relayed send records traffic but must not consume the rebind epoch. If it does, the next direct send to the
// relay host sees the epoch already current and never requeries, so the far side is never told to punch at our
// new address. This pins the SendVia call site, which the unit tests cannot reach.
func TestRebindRequeriesAfterRelayedSend(t *testing.T) {
t.Parallel()
ca, _, caKey, _ := cert_test.NewTestCaCert(cert.Version2, cert.Curve_CURVE25519, time.Now(), time.Now().Add(10*time.Minute), nil, nil, []string{})
// No lighthouse on purpose: it would hand out a direct address for them and nothing would relay.
myControl, myVpnIpNet, _, _ := newSimpleServer(cert.Version2, ca, caKey, "me", "10.128.0.1/24", m{"relay": m{"use_relays": true}})
relayControl, relayVpnIpNet, relayUdpAddr, _ := newSimpleServer(cert.Version2, ca, caKey, "relay", "10.128.0.128/24", m{"relay": m{"am_relay": true}})
theirControl, theirVpnIpNet, theirUdpAddr, _ := newSimpleServer(cert.Version2, ca, caKey, "them", "10.128.0.2/24", m{"relay": m{"use_relays": true}})
myControl.InjectLightHouseAddr(relayVpnIpNet[0].Addr(), relayUdpAddr)
myControl.InjectRelays(theirVpnIpNet[0].Addr(), []netip.Addr{relayVpnIpNet[0].Addr()})
relayControl.InjectLightHouseAddr(theirVpnIpNet[0].Addr(), theirUdpAddr)
r := router.NewR(t, myControl, relayControl, theirControl)
defer r.RenderFlow()
myControl.Start()
relayControl.Start()
theirControl.Start()
myControl.InjectTunPacket(BuildTunUDPPacket(theirVpnIpNet[0].Addr(), 80, myVpnIpNet[0].Addr(), 80, []byte("establish")))
r.RouteForAllUntilTxTun(theirControl)
r.RouteFor(time.Millisecond * 500)
hi := myControl.GetHostInfoByVpnAddr(theirVpnIpNet[0].Addr(), false)
require.NotNil(t, hi, "expected a tunnel to them")
require.NotEmpty(t, hi.CurrentRelaysToMe, "them must be reachable only via the relay for this test to mean anything")
// sendNoMetrics only reaches SendVia when there is no direct remote, so pin that too. Without this the test
// keeps passing while quietly sending direct and never exercising the relay path.
require.False(t, hi.CurrentRemote.IsValid(), "them must have no direct remote, otherwise SendVia is never called")
before, ok := myControl.GetRebindEpochFor(relayVpnIpNet[0].Addr())
require.True(t, ok, "expected a tunnel to the relay")
myControl.RebindUDPServer()
// Traffic to them goes through SendVia on the relay tunnel. That must record traffic without consuming the
// relay tunnel's own epoch edge, which belongs to the direct path.
myControl.InjectTunPacket(BuildTunUDPPacket(theirVpnIpNet[0].Addr(), 80, myVpnIpNet[0].Addr(), 80, []byte("relayed")))
r.RouteForAllUntilTxTun(theirControl)
after, ok := myControl.GetRebindEpochFor(relayVpnIpNet[0].Addr())
require.True(t, ok)
assert.Equal(t, before, after,
"a relayed send consumed the relay tunnel's rebind epoch, so the next direct send will not requery")
myControl.Stop()
relayControl.Stop()
theirControl.Stop()
}
+1
View File
@@ -22,6 +22,7 @@ require (
github.com/stefanberger/go-pkcs11uri v0.0.0-20230803200340-78284954bff6 github.com/stefanberger/go-pkcs11uri v0.0.0-20230803200340-78284954bff6
github.com/stretchr/testify v1.11.1 github.com/stretchr/testify v1.11.1
github.com/vishvananda/netlink v1.3.1 github.com/vishvananda/netlink v1.3.1
github.com/wlynxg/anet v0.0.5
go.uber.org/goleak v1.3.0 go.uber.org/goleak v1.3.0
go.yaml.in/yaml/v3 v3.0.4 go.yaml.in/yaml/v3 v3.0.4
golang.org/x/crypto v0.54.0 golang.org/x/crypto v0.54.0
+2
View File
@@ -149,6 +149,8 @@ github.com/vishvananda/netlink v1.3.1 h1:3AEMt62VKqz90r0tmNhog0r/PpWKmrEShJU0wJW
github.com/vishvananda/netlink v1.3.1/go.mod h1:ARtKouGSTGchR8aMwmkzC0qiNPrrWO5JS/XMVl45+b4= github.com/vishvananda/netlink v1.3.1/go.mod h1:ARtKouGSTGchR8aMwmkzC0qiNPrrWO5JS/XMVl45+b4=
github.com/vishvananda/netns v0.0.5 h1:DfiHV+j8bA32MFM7bfEunvT8IAqQ/NzSJHtcmW5zdEY= github.com/vishvananda/netns v0.0.5 h1:DfiHV+j8bA32MFM7bfEunvT8IAqQ/NzSJHtcmW5zdEY=
github.com/vishvananda/netns v0.0.5/go.mod h1:SpkAiCQRtJ6TvvxPnOSyH3BMl6unz3xZlaprSwhNNJM= github.com/vishvananda/netns v0.0.5/go.mod h1:SpkAiCQRtJ6TvvxPnOSyH3BMl6unz3xZlaprSwhNNJM=
github.com/wlynxg/anet v0.0.5 h1:J3VJGi1gvo0JwZ/P1/Yc/8p63SoW98B5dHkYDmpgvvU=
github.com/wlynxg/anet v0.0.5/go.mod h1:eay5PRQr7fIVAMbTbchTnO9gG65Hg/uYGdc7mguHxoA=
github.com/yuin/goldmark v1.1.27/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= github.com/yuin/goldmark v1.1.27/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74=
github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74=
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto= go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
+40 -79
View File
@@ -238,26 +238,18 @@ const (
) )
type HostInfo struct { type HostInfo struct {
// The first cache line is everything the packet paths touch.
remote atomic.Pointer[netip.AddrPort] remote atomic.Pointer[netip.AddrPort]
remotes *RemoteList
promoteCounter atomic.Uint32
ConnectionState *ConnectionState ConnectionState *ConnectionState
remoteIndexId uint32
// Traffic bits, pendingDeletion, and the rebind epoch we last sent under localIndexId uint32
state atomic.Uint32
promoteCounter atomic.Uint32
remoteIndexId uint32
localIndexId uint32
remotes *RemoteList
// vpnAddrs is a list of vpn addresses assigned to this host that are within our own vpn networks // vpnAddrs is a list of vpn addresses assigned to this host that are within our own vpn networks
// The host may have other vpn addresses that are outside our // The host may have other vpn addresses that are outside our
// vpn networks but were removed because they are not usable // vpn networks but were removed because they are not usable
vpnAddrs []netip.Addr vpnAddrs []netip.Addr
// Everything below is off the packet path: handshakes, relays, roaming and the connection manager.
// networks is a combination of specific vpn addresses (not prefixes!) and full unsafe networks assigned to this host. // networks is a combination of specific vpn addresses (not prefixes!) and full unsafe networks assigned to this host.
networks *bart.Table[NetworkType] networks *bart.Table[NetworkType]
relayState RelayState relayState RelayState
@@ -270,6 +262,11 @@ type HostInfo struct {
// This is used to limit lighthouse re-queries in chatty clients // This is used to limit lighthouse re-queries in chatty clients
nextLHQuery atomic.Int64 nextLHQuery atomic.Int64
// lastRebindCount is the other side of Interface.rebindCount, if these values don't match then we need to ask LH
// for a punch from the remote end of this tunnel. The goal being to prime their conntrack for our traffic just like
// with a handshake
lastRebindCount int8
// lastHandshakeTime records the time the remote side told us about at the stage when the handshake was completed locally // lastHandshakeTime records the time the remote side told us about at the stage when the handshake was completed locally
// Stage 1 packet will contain it if I am a responder, stage 2 packet if I am an initiator // Stage 1 packet will contain it if I am a responder, stage 2 packet if I am an initiator
// This is used to avoid an attack where a handshake packet is replayed after some time // This is used to avoid an attack where a handshake packet is replayed after some time
@@ -278,6 +275,9 @@ type HostInfo struct {
lastRoam time.Time lastRoam time.Time
lastRoamRemote netip.AddrPort lastRoamRemote netip.AddrPort
//TODO: in, out, and others might benefit from being an atomic.Int32. We could collapse connectionManager pendingDeletion, relayUsed, and in/out into this 1 thing
in, out, pendingDeletion atomic.Bool
// lastUsed tracks the last time ConnectionManager checked the tunnel and it was in use. // lastUsed tracks the last time ConnectionManager checked the tunnel and it was in use.
// This value will be behind against actual tunnel utilization in the hot path. // This value will be behind against actual tunnel utilization in the hot path.
// This should only be used by the ConnectionManagers ticker routine. // This should only be used by the ConnectionManagers ticker routine.
@@ -658,7 +658,7 @@ func (hm *HostMap) unlockedAddHostInfo(hostinfo *HostInfo, f *Interface) {
hm.Indexes[hostinfo.localIndexId] = hostinfo hm.Indexes[hostinfo.localIndexId] = hostinfo
hm.RemoteIndexes[hostinfo.remoteIndexId] = hostinfo hm.RemoteIndexes[hostinfo.remoteIndexId] = hostinfo
hostinfo.markOut(f.rebindEpoch.Load()) hostinfo.out.Store(true)
if f.connectionManager != nil { // f.connectionManager is only nil in some unit tests if f.connectionManager != nil { // f.connectionManager is only nil in some unit tests
f.connectionManager.trafficTimer.Add(hostinfo.localIndexId, f.connectionManager.checkInterval) f.connectionManager.trafficTimer.Add(hostinfo.localIndexId, f.connectionManager.checkInterval)
} }
@@ -759,68 +759,6 @@ func (i *HostInfo) TryPromoteBest(preferredRanges []netip.Prefix, ifce *Interfac
} }
} }
// Bits within HostInfo.state, everything above stateEpochShift is the epoch
const (
stateIn uint32 = 1 << iota
stateOut
statePendingDeletion
stateFlags = stateIn | stateOut | statePendingDeletion
stateEpochShift = 3
)
// markIn records inbound traffic
func (i *HostInfo) markIn() {
if i.state.Load()&stateIn == 0 {
i.state.Or(stateIn)
}
}
// markOut records a send and reports whether the epoch moved, meaning we want a punch from the far side
func (i *HostInfo) markOut(epoch uint32) bool {
e := epoch << stateEpochShift
for {
old := i.state.Load()
if old&stateOut != 0 && old&^stateFlags == e {
return false
}
if i.state.CompareAndSwap(old, old&stateFlags|stateOut|e) {
return old&^stateFlags != e
}
}
}
// markOutOnly records a send without consuming the rebind epoch, for paths that cannot act on a requery
func (i *HostInfo) markOutOnly() {
if i.state.Load()&stateOut == 0 {
i.state.Or(stateOut)
}
}
// sentSinceCheck reports whether anything has been sent since the connection manager last looked
func (i *HostInfo) sentSinceCheck() bool {
return i.state.Load()&stateOut != 0
}
// takeTraffic clears both traffic bits, leaving the epoch alone, and reports what they were
func (i *HostInfo) takeTraffic() (in bool, out bool) {
old := i.state.And(^(stateIn | stateOut))
return old&stateIn != 0, old&stateOut != 0
}
func (i *HostInfo) setPendingDeletion(v bool) {
if v {
i.state.Or(statePendingDeletion)
} else {
i.state.And(^statePendingDeletion)
}
}
func (i *HostInfo) isPendingDeletion() bool {
return i.state.Load()&statePendingDeletion != 0
}
func (i *HostInfo) GetCert() *cert.CachedCertificate { func (i *HostInfo) GetCert() *cert.CachedCertificate {
if i.ConnectionState != nil { if i.ConnectionState != nil {
return i.ConnectionState.peerCert return i.ConnectionState.peerCert
@@ -930,10 +868,28 @@ func (i *HostInfo) logger(l *slog.Logger) *slog.Logger {
// Utility functions // Utility functions
func localAddrs(l *slog.Logger, allowList *LocalAllowList) []netip.Addr { func localAddrs(l *slog.Logger, allowList *LocalAllowList) ([]netip.Addr, error) {
return collectLocalAddrs(l, allowList, localInterfaces, localInterfaceAddrs)
}
// collectLocalAddrs takes its enumerators as arguments so tests can drive the filtering and the
// failure branches without depending on the addresses of whatever host they run on. It reports
// failures to the caller rather than logging them, because it runs on every lighthouse update and
// only the caller can tell a new failure from a repeat of the same one.
func collectLocalAddrs(
l *slog.Logger,
allowList *LocalAllowList,
interfaces func() ([]net.Interface, error),
interfaceAddrs func(*net.Interface) ([]net.Addr, error),
) ([]netip.Addr, error) {
//FIXME: This function is pretty garbage //FIXME: This function is pretty garbage
var finalAddrs []netip.Addr var finalAddrs []netip.Addr
ifaces, _ := net.Interfaces() var errs []error
ifaces, err := interfaces()
if err != nil {
return nil, fmt.Errorf("failed to enumerate local interfaces: %w", err)
}
for _, i := range ifaces { for _, i := range ifaces {
allow := allowList.AllowName(i.Name) allow := allowList.AllowName(i.Name)
if l.Enabled(context.Background(), logging.LevelTrace) { if l.Enabled(context.Background(), logging.LevelTrace) {
@@ -946,7 +902,12 @@ func localAddrs(l *slog.Logger, allowList *LocalAllowList) []netip.Addr {
if !allow { if !allow {
continue continue
} }
addrs, _ := i.Addrs() addrs, err := interfaceAddrs(&i)
if err != nil {
errs = append(errs, fmt.Errorf("failed to get addresses for %s: %w", i.Name, err))
continue
}
for _, rawAddr := range addrs { for _, rawAddr := range addrs {
var addr netip.Addr var addr netip.Addr
switch v := rawAddr.(type) { switch v := rawAddr.(type) {
@@ -981,5 +942,5 @@ func localAddrs(l *slog.Logger, allowList *LocalAllowList) []netip.Addr {
} }
} }
} }
return finalAddrs return finalAddrs, errors.Join(errs...)
} }
+76 -34
View File
@@ -1,6 +1,8 @@
package nebula package nebula
import ( import (
"errors"
"net"
"net/netip" "net/netip"
"slices" "slices"
"testing" "testing"
@@ -402,42 +404,82 @@ func TestHostMap_RelayState(t *testing.T) {
} }
func TestHostInfo_markOut(t *testing.T) { func TestCollectLocalAddrs(t *testing.T) {
h := &HostInfo{} ifaces := []net.Interface{
h.markOut(5) // stamped when the tunnel was added {Index: 1, Name: "lo"},
{Index: 2, Name: "eth0"},
{Index: 3, Name: "docker0"},
}
addrs := map[string][]net.Addr{
"lo": {
&net.IPNet{IP: net.ParseIP("127.0.0.1"), Mask: net.CIDRMask(8, 32)},
&net.IPNet{IP: net.ParseIP("::1"), Mask: net.CIDRMask(128, 128)},
},
"eth0": {
&net.IPNet{IP: net.ParseIP("10.0.0.5"), Mask: net.CIDRMask(24, 32)},
&net.IPNet{IP: net.ParseIP("fe80::1"), Mask: net.CIDRMask(64, 128)},
&net.IPAddr{IP: net.ParseIP("fd00::5")},
},
"docker0": {
&net.IPNet{IP: net.ParseIP("172.17.0.1"), Mask: net.CIDRMask(16, 32)},
},
}
// A tunnel already on the current epoch has nothing to report, which is what keeps a fresh tunnel from enumerate := func() ([]net.Interface, error) { return ifaces, nil }
// requerying on its first packet addrsFor := func(i *net.Interface) ([]net.Addr, error) { return addrs[i.Name], nil }
assert.False(t, h.markOut(5), "an unchanged epoch should not report a move")
assert.True(t, h.sentSinceCheck(), "the send is still recorded as traffic")
// A rebind is observed exactly once, so we requery once per rebind // Loopback and link local are dropped, everything else on every interface is kept.
assert.True(t, h.markOut(6), "a bumped epoch should report a move") out, err := collectLocalAddrs(test.NewLogger(), nil, enumerate, addrsFor)
assert.False(t, h.markOut(6), "the epoch move should only be reported once") require.NoError(t, err)
assert.Equal(t, []netip.Addr{
netip.MustParseAddr("10.0.0.5"),
netip.MustParseAddr("fd00::5"),
netip.MustParseAddr("172.17.0.1"),
}, out)
// Traffic and pendingDeletion live in the same word and must survive an epoch change // An interface the allow list rejects by name is never asked for its addresses.
h.setPendingDeletion(true) c := config.NewC(test.NewLogger())
h.markIn() c.Settings["allowlist"] = map[string]any{
assert.True(t, h.markOut(7)) "interfaces": map[string]any{`docker.*`: false},
assert.True(t, h.isPendingDeletion(), "pendingDeletion must survive an epoch change") }
in, out := h.takeTraffic() al, err := NewLocalAllowListFromConfig(c, "allowlist")
assert.True(t, in, "inbound traffic must survive an epoch change") require.NoError(t, err)
assert.True(t, out)
// Clearing the traffic bits leaves the epoch alone, otherwise an idle tunnel would requery forever asked := make(map[string]struct{})
assert.False(t, h.markOut(7), "takeTraffic must not disturb the epoch") countingAddrsFor := func(i *net.Interface) ([]net.Addr, error) {
} asked[i.Name] = struct{}{}
return addrs[i.Name], nil
// A relayed send records traffic but must leave the rebind epoch for the direct path to consume, otherwise }
// relaying to a host swallows the requery that gets the far side punching at our new address. out, err = collectLocalAddrs(test.NewLogger(), al, enumerate, countingAddrsFor)
func TestHostInfo_markOutOnly(t *testing.T) { require.NoError(t, err)
h := &HostInfo{} assert.Equal(t, []netip.Addr{
h.markOut(5) netip.MustParseAddr("10.0.0.5"),
netip.MustParseAddr("fd00::5"),
h.markOutOnly() }, out)
assert.True(t, h.sentSinceCheck(), "a relayed send is still outbound traffic") assert.NotContains(t, asked, "docker0")
assert.False(t, h.markOut(5), "a relayed send must not disturb the epoch")
// A failure to enumerate interfaces at all is reported rather than silently advertising nothing.
assert.True(t, h.markOut(6), "a relayed send must not consume the epoch edge") out, err = collectLocalAddrs(
assert.False(t, h.markOut(6)) test.NewLogger(),
nil,
func() ([]net.Interface, error) { return nil, errors.New("netlinkrib: permission denied") },
addrsFor,
)
assert.Nil(t, out)
require.EqualError(t, err, "failed to enumerate local interfaces: netlinkrib: permission denied")
// One interface failing is reported and skipped, the rest are still collected.
out, err = collectLocalAddrs(
test.NewLogger(),
nil,
enumerate,
func(i *net.Interface) ([]net.Addr, error) {
if i.Name == "eth0" {
return nil, errors.New("nope")
}
return addrs[i.Name], nil
},
)
assert.Equal(t, []netip.Addr{netip.MustParseAddr("172.17.0.1")}, out)
require.EqualError(t, err, "failed to get addresses for eth0: nope")
} }
+10 -4
View File
@@ -297,7 +297,7 @@ func (f *Interface) SendVia(via *HostInfo,
c := via.ConnectionState.messageCounter.Add(1) c := via.ConnectionState.messageCounter.Add(1)
out = header.Encode(out, header.Version, header.Message, header.MessageRelay, relay.RemoteIndex, c) out = header.Encode(out, header.Version, header.Message, header.MessageRelay, relay.RemoteIndex, c)
f.connectionManager.OutRelay(via) f.connectionManager.Out(via)
// Authenticate the header and payload, but do not encrypt for this message type. // Authenticate the header and payload, but do not encrypt for this message type.
// The payload consists of the inner, unencrypted Nebula header, as well as the end-to-end encrypted payload. // The payload consists of the inner, unencrypted Nebula header, as well as the end-to-end encrypted payload.
@@ -365,11 +365,17 @@ func (f *Interface) sendNoMetrics(t header.MessageType, st header.MessageSubType
//l.WithField("trace", string(debug.Stack())).Error("out Header ", &Header{Version, t, st, 0, hostinfo.remoteIndexId, c}, p) //l.WithField("trace", string(debug.Stack())).Error("out Header ", &Header{Version, t, st, 0, hostinfo.remoteIndexId, c}, p)
out = header.Encode(out, header.Version, t, st, hostinfo.remoteIndexId, c) out = header.Encode(out, header.Version, t, st, hostinfo.remoteIndexId, c)
// We rebound since this tunnel last sent, ask the lighthouse to get the far side punching at us again f.connectionManager.Out(hostinfo)
if f.connectionManager.Out(hostinfo) && t != header.CloseTunnel {
// Query our LH if we haven't since the last time we've been rebound, this will cause the remote to punch against
// all our addrs and enable a faster roaming.
if t != header.CloseTunnel && hostinfo.lastRebindCount != f.rebindCount {
//NOTE: there is an update hole if a tunnel isn't used and exactly 256 rebinds occur before the tunnel is
// finally used again. This tunnel would eventually be torn down and recreated if this action didn't help.
f.lightHouse.QueryServer(hostinfo.vpnAddrs[0]) f.lightHouse.QueryServer(hostinfo.vpnAddrs[0])
hostinfo.lastRebindCount = f.rebindCount
if f.l.Enabled(context.Background(), slog.LevelDebug) { if f.l.Enabled(context.Background(), slog.LevelDebug) {
f.l.Debug("Lighthouse update triggered for punch due to rebind epoch", f.l.Debug("Lighthouse update triggered for punch due to rebind counter",
"vpnAddrs", hostinfo.vpnAddrs, "vpnAddrs", hostinfo.vpnAddrs,
) )
} }
+2 -2
View File
@@ -82,8 +82,8 @@ type Interface struct {
sendRecvErrorConfig recvErrorConfig sendRecvErrorConfig recvErrorConfig
acceptRecvErrorConfig recvErrorConfig acceptRecvErrorConfig recvErrorConfig
// Bumped on every udp rebind, tunnels compare it to decide they need a punch from the far side // rebindCount is used to decide if an active tunnel should trigger a punch notification through a lighthouse
rebindEpoch atomic.Uint32 rebindCount int8
version string version string
conntrackCacheTimeout time.Duration conntrackCacheTimeout time.Duration
+27 -1
View File
@@ -40,6 +40,10 @@ type LightHouse struct {
// addresses rather than whatever this machine's NICs happen to be. Set it before Start. // addresses rather than whatever this machine's NICs happen to be. Set it before Start.
localAddrsFn func(*LocalAllowList) []netip.Addr localAddrsFn func(*LocalAllowList) []netip.Addr
// lastLocalAddrsErr is the previous localAddrsFn failure. Enumeration runs on every update, so an
// unchanged failure is demoted to Debug rather than warning every lighthouse.interval forever.
lastLocalAddrsErr atomic.Pointer[string]
// Local cache of answers from light houses // Local cache of answers from light houses
// map of vpn addr to answers // map of vpn addr to answers
addrMap map[netip.Addr]*RemoteList addrMap map[netip.Addr]*RemoteList
@@ -112,7 +116,9 @@ func NewLightHouseFromConfig(ctx context.Context, l *slog.Logger, c *config.C, c
l: l, l: l,
} }
h.localAddrsFn = func(al *LocalAllowList) []netip.Addr { h.localAddrsFn = func(al *LocalAllowList) []netip.Addr {
return localAddrs(h.l, al) addrs, err := localAddrs(h.l, al)
h.logLocalAddrsErr(err)
return addrs
} }
lighthouses := make([]netip.Addr, 0) lighthouses := make([]netip.Addr, 0)
@@ -913,6 +919,26 @@ func (lh *LightHouse) TriggerUpdate() {
} }
} }
// logLocalAddrsErr reports a localAddrs failure at Warn the first time it is seen and at Debug while
// it persists unchanged, so a permanent failure does not warn on every update forever.
func (lh *LightHouse) logLocalAddrsErr(err error) {
if err == nil {
lh.lastLocalAddrsErr.Store(nil)
return
}
msg := err.Error()
prev := lh.lastLocalAddrsErr.Swap(&msg)
if prev != nil && *prev == msg {
if lh.l.Enabled(context.Background(), slog.LevelDebug) {
lh.l.Debug("Failed to collect local addresses to advertise", "error", err)
}
return
}
lh.l.Warn("Failed to collect local addresses to advertise", "error", err)
}
func (lh *LightHouse) SendUpdate() { func (lh *LightHouse) SendUpdate() {
var v4 []*V4AddrPort var v4 []*V4AddrPort
var v6 []*V6AddrPort var v6 []*V6AddrPort
+31
View File
@@ -1,7 +1,9 @@
package nebula package nebula
import ( import (
"bytes"
"encoding/binary" "encoding/binary"
"errors"
"fmt" "fmt"
"net/netip" "net/netip"
"testing" "testing"
@@ -738,3 +740,32 @@ func TestLighthouse_DeletesWork(t *testing.T) {
out = lh.Query(testHost) out = lh.Query(testHost)
assert.Nil(t, out) assert.Nil(t, out)
} }
func TestLightHouse_logLocalAddrsErr(t *testing.T) {
out := &bytes.Buffer{}
lh := &LightHouse{l: test.NewLoggerWithOutput(out)}
// The first sighting of a failure warns.
lh.logLocalAddrsErr(errors.New("permission denied"))
assert.Contains(t, out.String(), "level=WARN")
assert.Contains(t, out.String(), "permission denied")
// Repeating unchanged does not warn again, which is what keeps a permanent failure from warning
// on every lighthouse.interval for the life of the process.
out.Reset()
lh.logLocalAddrsErr(errors.New("permission denied"))
assert.NotContains(t, out.String(), "level=WARN")
// A different failure is a new event and warns.
out.Reset()
lh.logLocalAddrsErr(errors.New("something else"))
assert.Contains(t, out.String(), "level=WARN")
assert.Contains(t, out.String(), "something else")
// Recovering resets, so the same failure returning later warns again.
out.Reset()
lh.logLocalAddrsErr(nil)
assert.Empty(t, out.String())
lh.logLocalAddrsErr(errors.New("something else"))
assert.Contains(t, out.String(), "level=WARN")
}
+13
View File
@@ -0,0 +1,13 @@
//go:build !android
package nebula
import "net"
func localInterfaces() ([]net.Interface, error) {
return net.Interfaces()
}
func localInterfaceAddrs(i *net.Interface) ([]net.Addr, error) {
return i.Addrs()
}
+32
View File
@@ -0,0 +1,32 @@
//go:build android
package nebula
import (
"net"
"github.com/wlynxg/anet"
)
// anet relies on //go:linkname and so needs -ldflags=-checklinkname=0 on Go 1.23+. Nebula ships no
// Android binaries of its own, so that burden falls on consumers linking Android artifacts.
func init() {
// anet only takes its bind-free path when it believes it is on API 30+, and detecting the running
// device's level requires cgo. Pin it so a CGO_ENABLED=0 build cannot quietly fall back to the
// denied path. The bind-free path is correct on older releases too, just unnecessary there.
anet.SetAndroidVersion(11)
}
// The app sandbox denies bind() on netlink_route_socket, so the stdlib's RTM_GETLINK enumeration
// fails with EACCES and we advertise no underlay addresses at all. anet reads RTM_GETADDR from an
// unbound socket instead, so this must not be collapsed back into net.Interfaces.
func localInterfaces() ([]net.Interface, error) {
return anet.Interfaces()
}
// net.Interface.Addrs goes back through the denied netlink path, so addresses have to come from anet
// as well. anet cannot report HardwareAddr, which localAddrs does not read.
func localInterfaceAddrs(i *net.Interface) ([]net.Addr, error) {
return anet.InterfaceAddrsByInterface(i)
}