mirror of
https://github.com/slackhq/nebula.git
synced 2026-09-30 17:36:38 +02:00
Compare commits
16
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4782c5da9c | ||
|
|
be77501f14 | ||
|
|
af439dadbf | ||
|
|
53c565eb29 | ||
|
|
4936b40169 | ||
|
|
06621e622e | ||
|
|
a304f370f4 | ||
|
|
e3efe0c53d | ||
|
|
8c5c740571 | ||
|
|
fb20de39b2 | ||
|
|
8b914e67f4 | ||
|
|
a4c813fd26 | ||
|
|
11528c6e02 | ||
|
|
192d994273 | ||
|
|
2c9f47c252 | ||
|
|
0ae7ba073c |
@@ -51,19 +51,15 @@ wsl -d $Distro -- bash -c "rm -rf $WslDir && mkdir -p $WslDir" | Out-Null
|
||||
$DevName = 'nebula-smoke'
|
||||
$Ip1 = '192.168.241.1'
|
||||
$Ip2 = '192.168.241.2'
|
||||
# Dual stack on purpose: a v4-only overlay never exercises the v6 side of tun.mtu.
|
||||
$Ip6_1 = 'fd42:4242:241::1'
|
||||
$Ip6_2 = 'fd42:4242:241::2'
|
||||
$Mtu = 1300
|
||||
$Port = 4242
|
||||
|
||||
& $NebulaCert ca -name 'smoke-ca' -out-crt "$WorkDir\ca.crt" -out-key "$WorkDir\ca.key"
|
||||
if ($LASTEXITCODE -ne 0) { throw "nebula-cert ca failed (exit $LASTEXITCODE)" }
|
||||
|
||||
& $NebulaCert sign -name 'lighthouse' -networks "$Ip1/24,$Ip6_1/64" -ca-crt "$WorkDir\ca.crt" -ca-key "$WorkDir\ca.key" -out-crt "$WorkDir\lighthouse.crt" -out-key "$WorkDir\lighthouse.key"
|
||||
& $NebulaCert sign -name 'lighthouse' -networks "$Ip1/24" -ca-crt "$WorkDir\ca.crt" -ca-key "$WorkDir\ca.key" -out-crt "$WorkDir\lighthouse.crt" -out-key "$WorkDir\lighthouse.key"
|
||||
if ($LASTEXITCODE -ne 0) { throw "nebula-cert sign lighthouse failed (exit $LASTEXITCODE)" }
|
||||
|
||||
& $NebulaCert sign -name 'peer' -networks "$Ip2/24,$Ip6_2/64" -ca-crt "$WorkDir\ca.crt" -ca-key "$WorkDir\ca.key" -out-crt "$WorkDir\peer.crt" -out-key "$WorkDir\peer.key"
|
||||
& $NebulaCert sign -name 'peer' -networks "$Ip2/24" -ca-crt "$WorkDir\ca.crt" -ca-key "$WorkDir\ca.key" -out-crt "$WorkDir\peer.crt" -out-key "$WorkDir\peer.key"
|
||||
if ($LASTEXITCODE -ne 0) { throw "nebula-cert sign peer failed (exit $LASTEXITCODE)" }
|
||||
|
||||
# Windows lighthouse config.
|
||||
@@ -86,7 +82,7 @@ tun:
|
||||
drop_local_broadcast: false
|
||||
drop_multicast: false
|
||||
tx_queue: 500
|
||||
mtu: $Mtu
|
||||
mtu: 1300
|
||||
network_category: private
|
||||
logging:
|
||||
level: info
|
||||
@@ -130,7 +126,7 @@ tun:
|
||||
drop_local_broadcast: false
|
||||
drop_multicast: false
|
||||
tx_queue: 500
|
||||
mtu: $Mtu
|
||||
mtu: 1300
|
||||
logging:
|
||||
level: info
|
||||
format: text
|
||||
@@ -173,7 +169,7 @@ Write-Host '=== WSL diagnostic ==='
|
||||
wsl --version 2>&1 | Out-Host
|
||||
wsl --list --verbose 2>&1 | Out-Host
|
||||
wsl -d $Distro -u root -- uname -a | Out-Host
|
||||
wsl -d $Distro -u root -- bash -c "modprobe tun 2>&1 || true; mkdir -p /dev/net; [ -c /dev/net/tun ] || mknod /dev/net/tun c 10 200; chmod 600 /dev/net/tun; { echo 0 > /proc/sys/net/ipv6/conf/all/disable_ipv6; echo 0 > /proc/sys/net/ipv6/conf/default/disable_ipv6; } 2>/dev/null || true; ls -l /dev/net/tun"
|
||||
wsl -d $Distro -u root -- bash -c "modprobe tun 2>&1 || true; mkdir -p /dev/net; [ -c /dev/net/tun ] || mknod /dev/net/tun c 10 200; chmod 600 /dev/net/tun; ls -l /dev/net/tun"
|
||||
if ($LASTEXITCODE -ne 0) { throw "failed to prepare /dev/net/tun in WSL (TUN support missing?)" }
|
||||
|
||||
# Deliberately no New-NetFirewallRule calls here -- nebula's windows_bypass_wdf
|
||||
@@ -218,16 +214,6 @@ try {
|
||||
}
|
||||
Write-Host "OK: $DevName NetworkCategory=Private"
|
||||
|
||||
# v6 silently kept the adapter default of 65535 while v4 was correct.
|
||||
foreach ($family in @('IPv4', 'IPv6')) {
|
||||
Wait-Until -TimeoutSec 30 -What "$DevName $family NlMtu=$Mtu" -Predicate {
|
||||
if ($lhProc.HasExited) { throw "lighthouse exited (code $($lhProc.ExitCode)) before $family mtu was set" }
|
||||
$rows = @(Get-NetIPInterface -InterfaceAlias $DevName -AddressFamily $family -ErrorAction SilentlyContinue)
|
||||
$rows.Count -gt 0 -and -not ($rows | Where-Object { $_.NlMtu -ne $Mtu })
|
||||
}
|
||||
Write-Host "OK: $DevName $family NlMtu=$Mtu"
|
||||
}
|
||||
|
||||
Wait-Until -TimeoutSec 30 -What "WSL nebula1 with $Ip2" -Predicate {
|
||||
if ($peerProc.HasExited) { throw "peer exited (code $($peerProc.ExitCode)) before tun was ready" }
|
||||
$r = wsl -d $Distro -u root -- bash -c "ip -o addr show nebula1 2>/dev/null | grep -q 'inet $Ip2' && echo yes"
|
||||
@@ -235,13 +221,6 @@ try {
|
||||
}
|
||||
Write-Host "OK: WSL nebula1 has $Ip2"
|
||||
|
||||
Wait-Until -TimeoutSec 30 -What "WSL nebula1 with $Ip6_2" -Predicate {
|
||||
if ($peerProc.HasExited) { throw "peer exited (code $($peerProc.ExitCode)) before the v6 address was up" }
|
||||
$r = wsl -d $Distro -u root -- bash -c "ip -o addr show nebula1 2>/dev/null | grep -q 'inet6 $Ip6_2' && echo yes"
|
||||
("$r").Trim() -eq 'yes'
|
||||
}
|
||||
Write-Host "OK: WSL nebula1 has $Ip6_2"
|
||||
|
||||
Wait-Until -TimeoutSec 30 -What "ping from WSL peer to windows lighthouse ($Ip1)" -Predicate {
|
||||
if ($peerProc.HasExited) { throw "peer exited (code $($peerProc.ExitCode)) before ping succeeded" }
|
||||
$r = wsl -d $Distro -u root -- bash -c "ping -c1 -W1 $Ip1 >/dev/null 2>&1 && echo OK"
|
||||
@@ -255,14 +234,6 @@ try {
|
||||
}
|
||||
Write-Host "OK: windows lighthouse -> WSL peer"
|
||||
|
||||
# Otherwise the v6 networks only prove the interface exists, not that it forwards.
|
||||
Wait-Until -TimeoutSec 30 -What "v6 ping from WSL peer to windows lighthouse ($Ip6_1)" -Predicate {
|
||||
if ($peerProc.HasExited) { throw "peer exited (code $($peerProc.ExitCode)) before the v6 ping succeeded" }
|
||||
$r = wsl -d $Distro -u root -- bash -c "ping -6 -c1 -W1 $Ip6_1 >/dev/null 2>&1 && echo OK"
|
||||
("$r").Trim() -eq 'OK'
|
||||
}
|
||||
Write-Host "OK: WSL peer -> windows lighthouse over v6"
|
||||
|
||||
Write-Host ''
|
||||
Write-Host 'All smoke checks passed.'
|
||||
}
|
||||
|
||||
+1
-14
@@ -7,10 +7,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [1.11.1] - 2026-08-21
|
||||
|
||||
See the [v1.11.1](https://github.com/slackhq/nebula/milestone/30?closed=1) milestone for a complete list of changes.
|
||||
|
||||
### Changed
|
||||
|
||||
- IPv6 packets whose next header is a protocol Nebula does not parse (SCTP, GRE, IP-in-IP, etc.) are now
|
||||
@@ -19,18 +15,11 @@ See the [v1.11.1](https://github.com/slackhq/nebula/milestone/30?closed=1) miles
|
||||
their true protocol, so only a `proto: any` rule allows them. If you carry one of these protocols over the
|
||||
overlay, confirm a `proto: any` rule covers it before upgrading, it may have been passing only through this
|
||||
bypass. (#1840)
|
||||
- Drop the dependency on `github.com/cyberdelia/go-metrics-graphite`, which has been unmaintained for over ten
|
||||
years, by inlining the small amount of code Nebula used. (#1832)
|
||||
|
||||
### Fixed
|
||||
|
||||
- The ICMPv6 type was read from the wrong byte when classifying IPv6 packets, so the echo identifier used
|
||||
for conntrack was never picked up. (#1840)
|
||||
- Enforce outbound message counter limits so a tunnel is rehandshaked before the counter can wrap, preventing
|
||||
nonce reuse. This is unreachable in practice, but is enforced as a defense-in-depth measure. (#1841)
|
||||
- Prevent `nebula-cert ca` from running out of memory on 32bit systems when generating encrypted private keys. (#1834)
|
||||
- Tolerate `ErrDumpInterrupted` when listing tun addresses on Linux, so a transient interrupted netlink dump
|
||||
no longer aborts startup. (#1835)
|
||||
|
||||
## [1.11.0] - 2026-07-23
|
||||
|
||||
@@ -895,9 +884,7 @@ created.)
|
||||
|
||||
- Initial public release.
|
||||
|
||||
[Unreleased]: https://github.com/slackhq/nebula/compare/v1.11.1...HEAD
|
||||
[1.11.1]: https://github.com/slackhq/nebula/releases/tag/v1.11.1
|
||||
[1.11.0]: https://github.com/slackhq/nebula/releases/tag/v1.11.0
|
||||
[Unreleased]: https://github.com/slackhq/nebula/compare/v1.10.3...HEAD
|
||||
[1.10.3]: https://github.com/slackhq/nebula/releases/tag/v1.10.3
|
||||
[1.10.2]: https://github.com/slackhq/nebula/releases/tag/v1.10.2
|
||||
[1.10.1]: https://github.com/slackhq/nebula/releases/tag/v1.10.1
|
||||
|
||||
+28
-3
@@ -197,6 +197,27 @@ func (cm *connectionManager) doTrafficCheck(localIndex uint32, p, nb, out []byte
|
||||
}
|
||||
|
||||
cm.resetRelayTrafficCheck(hostinfo)
|
||||
cm.maintainLanes(localIndex, decision, hostinfo, now, nb, out)
|
||||
}
|
||||
|
||||
// maintainLanes piggybacks multiport lane probing on the per-tunnel traffic
|
||||
// tick. This tick is the right place for it precisely because lanes are
|
||||
// demand-driven: a tunnel only lands here when it has traffic, which is the
|
||||
// same condition that raises lane demand.
|
||||
//
|
||||
// makeTrafficDecision returns a nil hostinfo on some keep-alive paths, so
|
||||
// re-resolve the index in that case.
|
||||
func (cm *connectionManager) maintainLanes(localIndex uint32, decision trafficDecision, hostinfo *HostInfo, now time.Time, nb, out []byte) {
|
||||
if decision == deleteTunnel || decision == closeTunnel {
|
||||
return
|
||||
}
|
||||
if hostinfo == nil {
|
||||
hostinfo = cm.hostMap.QueryIndex(localIndex)
|
||||
if hostinfo == nil {
|
||||
return
|
||||
}
|
||||
}
|
||||
cm.intf.probeLanes(hostinfo, now, nb, out)
|
||||
}
|
||||
|
||||
func (cm *connectionManager) resetRelayTrafficCheck(hostinfo *HostInfo) {
|
||||
@@ -329,7 +350,10 @@ func (cm *connectionManager) makeTrafficDecision(localIndex uint32, now time.Tim
|
||||
return closeTunnel, hostinfo, nil
|
||||
}
|
||||
|
||||
if hostinfo.ConnectionState != nil && hostinfo.ConnectionState.messageCounter.Load() >= RejectAfterMessages {
|
||||
// The highest counter across the base session and its lanes: the lanes carry
|
||||
// the data, so the base counter alone would sit near zero while a lane runs
|
||||
// its keys past the nonce ceiling.
|
||||
if hostinfo.maxMessageCounter() >= RejectAfterMessages {
|
||||
// Send path can't encrypt a CloseTunnel notify, so just delete locally; the peer recovers via recv_error.
|
||||
hostinfo.logger(cm.l).Error("Dropping tunnel, message counter is exhausted")
|
||||
return deleteTunnel, hostinfo, nil
|
||||
@@ -437,6 +461,7 @@ func (cm *connectionManager) isInactive(hostinfo *HostInfo, now time.Time) (time
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// Lane traffic is this hostinfo's traffic, so lastUsed already covers it.
|
||||
inactiveDuration := now.Sub(hostinfo.lastUsed)
|
||||
if inactiveDuration < cm.getInactivityTimeout() {
|
||||
// It's not considered inactive
|
||||
@@ -460,7 +485,7 @@ func (cm *connectionManager) shouldSwapPrimary(current *HostInfo) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
if current.ConnectionState.messageCounter.Load() >= RehandshakeAfterMessages {
|
||||
if current.maxMessageCounter() >= RehandshakeAfterMessages {
|
||||
// This tunnel is being rolled for counter exhaustion, never swap back onto its spent key.
|
||||
return false
|
||||
}
|
||||
@@ -564,7 +589,7 @@ func (cm *connectionManager) tryRehandshake(hostinfo *HostInfo) {
|
||||
cm.intf.handshakeManager.StartHandshake(hostinfo.vpnAddrs[0], nil)
|
||||
return
|
||||
}
|
||||
if hostinfo.ConnectionState.messageCounter.Load() >= RehandshakeAfterMessages {
|
||||
if hostinfo.maxMessageCounter() >= RehandshakeAfterMessages {
|
||||
cm.l.Info("Re-handshaking with remote",
|
||||
"vpnAddrs", hostinfo.vpnAddrs,
|
||||
"reason", "message counter rehandshake threshold reached",
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
package nebula
|
||||
|
||||
import (
|
||||
"crypto/hkdf"
|
||||
"crypto/sha256"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strconv"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
|
||||
@@ -74,6 +77,53 @@ func newConnectionStateFromResult(r *handshake.Result) (*ConnectionState, error)
|
||||
return ci, nil
|
||||
}
|
||||
|
||||
// newLaneConnectionState derives multiport lane s's session from the base
|
||||
// tunnel's material. Each key is an HKDF expansion of the base tunnel's matching
|
||||
// key, labelled with the lane index, so the pair stays matched with no extra
|
||||
// negotiation: Noise leaves our send key equal to the peer's receive key, and
|
||||
// expanding both with the same label preserves that.
|
||||
//
|
||||
// The lane gets its own counter and replay window starting from zero. No
|
||||
// handshake messages were spent on it, so unlike the base session there is
|
||||
// nothing to seed.
|
||||
func newLaneConnectionState(m *laneMaterial, lane uint8) (*ConnectionState, error) {
|
||||
if lane == 0 {
|
||||
return nil, fmt.Errorf("lane 0 is the base session")
|
||||
}
|
||||
|
||||
eKey, err := deriveLaneKey(m.eKey, lane)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
dKey, err := deriveLaneKey(m.dKey, lane)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return &ConnectionState{
|
||||
myCert: m.myCert,
|
||||
initiator: m.initiator,
|
||||
peerCert: m.peerCert,
|
||||
eKey: noiseutil.NewCipherStateFromKey(eKey, m.cipher),
|
||||
dKey: noiseutil.NewCipherStateFromKey(dKey, m.cipher),
|
||||
window: NewBits(ReplayWindow),
|
||||
epoch: sessionEpoch.Add(1),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// deriveLaneKey expands a base tunnel key into the key for one lane.
|
||||
func deriveLaneKey(base [32]byte, lane uint8) ([32]byte, error) {
|
||||
var out [32]byte
|
||||
// The base key is already unique to this tunnel and direction, so the lane
|
||||
// index is the only thing that needs to vary; no salt is required.
|
||||
k, err := hkdf.Key(sha256.New, base[:], nil, laneKeyInfo+" "+strconv.Itoa(int(lane)), len(out))
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
copy(out[:], k)
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (cs *ConnectionState) MarshalJSON() ([]byte, error) {
|
||||
return json.Marshal(m{
|
||||
"certificate": cs.peerCert,
|
||||
@@ -118,6 +168,15 @@ func (cs *ConnectionState) Decrypt(l *slog.Logger, messageCounter uint64, packet
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// noteSeen records a counter that some other session for the same keys already
|
||||
// accepted, so a packet doesn't become replayable just because the session that
|
||||
// decrypted it was thrown away. See laneSet.installSession, its only caller.
|
||||
func (cs *ConnectionState) noteSeen(l *slog.Logger, messageCounter uint64) {
|
||||
cs.decryptLock.Lock()
|
||||
cs.window.Update(l, messageCounter)
|
||||
cs.decryptLock.Unlock()
|
||||
}
|
||||
|
||||
func (cs *ConnectionState) VerifyRelay(l *slog.Logger, messageCounter uint64, packet []byte, nb []byte) error {
|
||||
cs.decryptLock.Lock()
|
||||
result := cs.window.Check(l, messageCounter)
|
||||
|
||||
@@ -58,6 +58,7 @@ func runTestHandshake(t *testing.T) (initR, respR *handshake.Result) {
|
||||
cert.Version2, initCreds, verifier,
|
||||
func() (uint32, error) { return 1000, nil },
|
||||
true, header.HandshakeIXPSK0,
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
|
||||
@@ -65,6 +66,7 @@ func runTestHandshake(t *testing.T) (initR, respR *handshake.Result) {
|
||||
cert.Version2, respCreds, verifier,
|
||||
func() (uint32, error) { return 2000, nil },
|
||||
false, header.HandshakeIXPSK0,
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
|
||||
|
||||
+42
-2
@@ -67,6 +67,16 @@ type ControlHostInfo struct {
|
||||
CurrentRemote netip.AddrPort `json:"currentRemote"`
|
||||
CurrentRelaysToMe []netip.Addr `json:"currentRelaysToMe"`
|
||||
CurrentRelaysThroughMe []netip.Addr `json:"currentRelaysThroughMe"`
|
||||
Lanes []ControlLane `json:"lanes,omitempty"`
|
||||
}
|
||||
|
||||
// ControlLane reports one multiport lane of a tunnel. Only lanes we may send on
|
||||
// are listed; receive-only lanes have no state worth showing.
|
||||
type ControlLane struct {
|
||||
Index uint8 `json:"index"`
|
||||
Up bool `json:"up"`
|
||||
Remote netip.AddrPort `json:"remote,omitempty"`
|
||||
MessageCounter uint64 `json:"messageCounter"`
|
||||
}
|
||||
|
||||
// Start actually runs nebula, this is a nonblocking call.
|
||||
@@ -202,10 +212,15 @@ func (c *Control) RebindUDPServer() {
|
||||
return
|
||||
}
|
||||
|
||||
// Every socket needs rebinding, not just the base: with multiport each one is bound to its own lane port, and
|
||||
// even without it the surplus SO_REUSEPORT sockets stay pinned to the interface we came up on otherwise.
|
||||
//
|
||||
// A failure here means we are likely still pinned to the interface we came up on, so the rest of this is
|
||||
// unlikely to help. Say so instead of silently carrying on as if we rebound.
|
||||
if err := c.f.outside.Rebind(); err != nil {
|
||||
c.l.Error("Failed to rebind udp socket", "error", err)
|
||||
for i, w := range c.f.writers {
|
||||
if err := w.Rebind(); err != nil {
|
||||
c.l.Error("Failed to rebind udp socket", "error", err, "writer", i)
|
||||
}
|
||||
}
|
||||
|
||||
// Trigger a lighthouse update, useful for mobile clients that should have an update interval of 0
|
||||
@@ -385,6 +400,7 @@ func copyHostInfo(h *HostInfo, preferredRanges []netip.Prefix) ControlHostInfo {
|
||||
CurrentRelaysToMe: h.relayState.CopyRelayIps(),
|
||||
CurrentRelaysThroughMe: h.relayState.CopyRelayForIps(),
|
||||
CurrentRemote: h.GetRemote(),
|
||||
Lanes: copyLanes(h),
|
||||
}
|
||||
|
||||
for i, a := range h.vpnAddrs {
|
||||
@@ -402,6 +418,30 @@ func copyHostInfo(h *HostInfo, preferredRanges []netip.Prefix) ControlHostInfo {
|
||||
return chi
|
||||
}
|
||||
|
||||
// copyLanes snapshots the sendable multiport lanes of a tunnel, or nil when it
|
||||
// has none. txAddr is the lane's gate as well as its destination, so a nil load
|
||||
// is exactly "this lane is down and its routine is riding the base tunnel".
|
||||
func copyLanes(h *HostInfo) []ControlLane {
|
||||
ls := h.lanes
|
||||
if ls == nil || ls.txLanes < 2 {
|
||||
return nil
|
||||
}
|
||||
|
||||
lanes := make([]ControlLane, 0, ls.txLanes-1)
|
||||
for s := 1; s < ls.txLanes; s++ {
|
||||
l := ControlLane{Index: uint8(s)}
|
||||
if addr := ls.txAddr[s].Load(); addr != nil {
|
||||
l.Up = true
|
||||
l.Remote = *addr
|
||||
}
|
||||
if cs := ls.sessions[s].Load(); cs != nil {
|
||||
l.MessageCounter = cs.messageCounter.Load()
|
||||
}
|
||||
lanes = append(lanes, l)
|
||||
}
|
||||
return lanes
|
||||
}
|
||||
|
||||
func listHostMapHosts(hl controlHostLister) []ControlHostInfo {
|
||||
hosts := make([]ControlHostInfo, 0)
|
||||
pr := hl.GetPreferredRanges()
|
||||
|
||||
+1
-1
@@ -105,7 +105,7 @@ func TestControl_GetHostInfoByVpnIp(t *testing.T) {
|
||||
}
|
||||
|
||||
// Make sure we don't have any unexpected fields
|
||||
assertFields(t, []string{"VpnAddrs", "LocalIndex", "RemoteIndex", "RemoteAddrs", "Cert", "MessageCounter", "CurrentRemote", "CurrentRelaysToMe", "CurrentRelaysThroughMe"}, thi)
|
||||
assertFields(t, []string{"VpnAddrs", "LocalIndex", "RemoteIndex", "RemoteAddrs", "Cert", "MessageCounter", "CurrentRemote", "CurrentRelaysToMe", "CurrentRelaysThroughMe", "Lanes"}, thi)
|
||||
assert.Equal(t, &expectedInfo, thi)
|
||||
test.AssertDeepCopyEqual(t, &expectedInfo, thi)
|
||||
|
||||
|
||||
@@ -0,0 +1,526 @@
|
||||
# Multiport lanes
|
||||
|
||||
Status: experimental. Linux only. Off unless `multiport.ports` is set.
|
||||
|
||||
## The problem
|
||||
|
||||
A nebula tunnel is one UDP 4-tuple. Everything between the two hosts that makes
|
||||
a decision per flow makes it once, for the whole tunnel:
|
||||
|
||||
- **ECMP / LAG** hashes the 4-tuple and picks one path. A tunnel gets one path's
|
||||
worth of bandwidth no matter how many exist.
|
||||
- **NIC RSS** hashes the 4-tuple to one receive queue, so one CPU takes every
|
||||
interrupt for the tunnel and receive is capped by a single core.
|
||||
- **Per-flow policers and shapers** see one flow and rate-limit it as one.
|
||||
- **Cloud per-flow bandwidth caps** are the hard version of that. AWS EC2 meters
|
||||
each 5-tuple separately and caps a single flow well below what the instance can
|
||||
do in aggregate -- on the order of 5 Gbps for a single flow within a VPC (more
|
||||
inside a cluster placement group, or with ENA Express; check the current EC2
|
||||
network-limits docs for exact numbers) on instances whose aggregate allowance
|
||||
is many times that. Other providers do the same. Nothing about the path is
|
||||
saturated when this fires, and no amount of retuning changes it: the hypervisor
|
||||
is metering *the flow*, so the only way to get more is to be more than one
|
||||
flow.
|
||||
|
||||
Running `routines: N` does not help. Those N sockets share one port through
|
||||
`SO_REUSEPORT`, and the group is keyed on the exact `(addr, port)` pair, so the
|
||||
kernel's reuseport hash is the *only* thing spreading the work -- every hash
|
||||
outside this host still sees a single flow.
|
||||
|
||||
Multiport gives one tunnel several underlay 4-tuples, so all of those per-flow
|
||||
decisions get made several times, independently. A tunnel with N lanes is N
|
||||
flows to everything counting flows: N ECMP hashes, N receive queues, N of the
|
||||
cloud provider's per-flow buckets.
|
||||
|
||||
There is a second bottleneck, and it is inside this host rather than out on the
|
||||
network: in FIPS 140 mode the AES-GCM implementation refuses to seal twice under
|
||||
the same nonce, which forces every encrypt on a session to be serialized behind
|
||||
one mutex. `routines: N` does not help there either. Because each lane is a
|
||||
separate session, multiport splits that mutex as well -- see
|
||||
[The FIPS 140 encrypt lock](#the-fips-140-encrypt-lock).
|
||||
|
||||
## What a lane is
|
||||
|
||||
A lane is **not** a second tunnel. It is an extra session on the same
|
||||
`HostInfo`.
|
||||
|
||||
Noise leaves both sides with `A.eKey == B.dKey` and `A.dKey == B.eKey`. Expanding
|
||||
both keys through HKDF-SHA256 with the same per-lane label preserves that
|
||||
equality, so both sides land on a matched key pair having exchanged nothing:
|
||||
|
||||
```
|
||||
lane s send key = HKDF(base eKey, info: "nebula multiport lane v1 <s>")
|
||||
lane s recv key = HKDF(base dKey, info: "nebula multiport lane v1 <s>")
|
||||
```
|
||||
|
||||
(`connection_state.go:deriveLaneKey`. The base key is already unique per tunnel
|
||||
and per direction, so the lane index is the only thing that needs to vary and no
|
||||
salt is required.)
|
||||
|
||||
Consequences worth stating plainly:
|
||||
|
||||
- A lane costs **no handshake** and has no half-established state.
|
||||
- A lane dies exactly when its base tunnel dies. There is no independent
|
||||
lifetime to reason about, no second teardown path.
|
||||
- Each lane is a **full** session: its own message counter, its own replay
|
||||
window, its own cipher states, its own encrypt lock. Flows on different paths
|
||||
never contend for shared replay state, which is what makes reordering across
|
||||
lanes harmless, and under FIPS 140 the separate encrypt locks are what let one
|
||||
tunnel encrypt on more than one core.
|
||||
- Rolling the base tunnel replaces every lane key, because lanes are derived
|
||||
from it. `maxMessageCounter` therefore reports the max across the base and all
|
||||
lane counters, so rehandshake and exhaustion thresholds see the real data
|
||||
volume rather than the base session's small share.
|
||||
|
||||
Which lane a packet belongs to travels in the nebula header, in the low 8 bits
|
||||
of what used to be `Reserved` (see `header/header.go`). It is part of the AEAD's
|
||||
associated data, so a lane index cannot be altered in flight -- a packet
|
||||
decrypts on the lane it claims or not at all.
|
||||
|
||||
**Lane 0 is the base tunnel itself**: `HostInfo.ConnectionState`, the base port,
|
||||
the peer's real remote address. It is not a special case reserved for control
|
||||
traffic; it carries its share of data flows like any other lane.
|
||||
|
||||
## Socket layout
|
||||
|
||||
`routines` is **per port**. Each of `multiport.ports` consecutive ports gets its
|
||||
own full group of `routines` sockets sharing it through `SO_REUSEPORT`, so total
|
||||
sockets = total routines = total tun queues = `routines * multiport.ports`.
|
||||
|
||||
```
|
||||
routines: 2, multiport.ports: 3, listen.port: 4242
|
||||
|
||||
port 4242 port 4243 port 4244
|
||||
(lane 0/base) (lane 1) (lane 2)
|
||||
+-------------+ +-------------+ +-------------+
|
||||
writers[] | 0 | 1 | | 2 | 3 | | 4 | 5 |
|
||||
+-------------+ +-------------+ +-------------+
|
||||
routine 0 1 0 1 0 1
|
||||
```
|
||||
|
||||
Sockets are laid out **port-major**: `writers[s*routinesPerPort + r]` is the
|
||||
r'th socket on port `listen.port+s`. So:
|
||||
|
||||
```go
|
||||
laneSock(q, s) = s*routinesPerPort + q%routinesPerPort // inside.go
|
||||
egressSock(q) = laneSock(q, 0) // base traffic
|
||||
```
|
||||
|
||||
Two properties fall out of this, and the rest of the design depends on both:
|
||||
|
||||
1. **Every routine owns exactly one socket.** `listenIn` blocks in `recvmmsg`
|
||||
and everything downstream of it -- the batcher, the conntrack cache, the
|
||||
`txQueue` -- is single-owner and lock-free. More sockets than routines would
|
||||
need epoll or locks.
|
||||
2. **Every routine has a sibling socket at the same group position on every
|
||||
port.** So socket selection is a pure function of `(queue, lane)` with no
|
||||
borrowing and no shared state, and each port's traffic spreads across its
|
||||
whole group rather than funnelling into one socket.
|
||||
|
||||
The second property is why `routines` is per port rather than a total to divide
|
||||
up. With one socket per port, each port would be served by a single core -- worst
|
||||
of all the base port, which carries every handshake, every lighthouse and punch
|
||||
packet, every peer without multiport, and every tunnel whose lanes are down.
|
||||
|
||||
## Negotiation
|
||||
|
||||
Multiport capability rides the existing handshake payload as two new protobuf
|
||||
fields (`handshake/payload.go`): `InitiatorLanes` field 9 and `ResponderLanes`
|
||||
field 10, each a `LaneDetails{PortCount, BasePort, TxLanes}`.
|
||||
|
||||
- `PortCount` / `BasePort` -- the contiguous port range the sender bound, so the
|
||||
peer knows where to aim its lanes.
|
||||
- `TxLanes` -- how many lanes the sender may send on, so the peer knows how many
|
||||
lane sessions it must be prepared to receive on.
|
||||
|
||||
Both are `nil` when multiport is off, which keeps the encoded payload
|
||||
**byte-identical** to a vanilla one. A peer that has never heard of lanes skips
|
||||
unknown fields as protobuf requires, and gets a plain tunnel.
|
||||
|
||||
From the result, `newLaneSet` computes:
|
||||
|
||||
```go
|
||||
sessions = min(max(myLanes, peerTxLanes), 256) // we must be able to RECEIVE all of theirs
|
||||
txLanes = min(myLanes, peerPortCount, sessions) // we may only SEND on ports they bound
|
||||
```
|
||||
|
||||
Sizing RX by the peer's count and TX by our own is what lets asymmetric hosts
|
||||
work: a 4-port laptop talking to a 32-port server sends on 4 lanes and receives
|
||||
on 32.
|
||||
|
||||
### Port pairing
|
||||
|
||||
Lane `s` targets `peerBasePort + ((s + portOffset) % peerPortCount)`.
|
||||
|
||||
`portOffset` is a per-pair FNV hash of the sorted vpn-address pair. Without it,
|
||||
every small peer would aim its few lanes at a big peer's first few ports and
|
||||
concentrate that peer's receive work on a couple of sockets.
|
||||
|
||||
The rotation has to cancel, though, or the two directions of one flow would take
|
||||
unrelated 4-tuples and neither side's traffic would arrive through the conntrack
|
||||
or NAT entry the other's probe opened. So both sides hash the *same* sorted pair
|
||||
and the higher-addressed side **negates** the result. When the port counts match,
|
||||
the two rotations cancel exactly: our lane `s`'s 4-tuple is the reverse of the
|
||||
peer's lane `s`. `laneBias` does the matching rotation on the flow hash for the
|
||||
same reason.
|
||||
|
||||
That pairing only exists when our lane indices map one-to-one onto the peer's
|
||||
ports, which is exactly the `txLanes == peerPortCount` test `newLaneSet` applies
|
||||
before setting `laneBias`. If we send on 4 lanes and the peer bound 8 ports, four
|
||||
of its ports have no lane of ours pointing at them, and no choice of rotation can
|
||||
make our lane `s` and its lane `s` be each other's reverse. So in that case
|
||||
`laneBias` stays 0 and each side hashes the flow to a lane on its own. The flow
|
||||
still works and is still spread; the two directions simply take two unrelated
|
||||
4-tuples instead of one 4-tuple and its exact reverse, and each direction depends
|
||||
on its own lane's probe having opened its own conntrack or NAT entry.
|
||||
|
||||
## Bringing a lane up
|
||||
|
||||
Receiving on a lane needs no permission: the keys are derivable the moment the
|
||||
base handshake completes. **Sending** on one needs proof the new 4-tuple actually
|
||||
works, because nothing else would notice a middlebox quietly dropping it. So:
|
||||
|
||||
```
|
||||
lane down --probe (Test/LaneProbe on lane s, from port base+s to peer's lane port)-->
|
||||
<--ack (Test/LaneProbeAck, on the BASE tunnel)--
|
||||
lane up
|
||||
```
|
||||
|
||||
- The probe is encrypted with **the lane's own session**, so an ack proves the
|
||||
whole lane end to end: our source port reached the peer, its reply reached us,
|
||||
and the keys we derived match the ones it derived.
|
||||
- The ack rides the **base tunnel** on purpose. A probe proves the peer's lane
|
||||
works in *its* send direction; answering on our own lane `s` would make the
|
||||
result depend on a second path that can be broken independently.
|
||||
- The ack echoes the header's lane, not the payload's, so a peer cannot get us to
|
||||
vouch for a lane it did not probe.
|
||||
- A generation byte in the probe is echoed in the ack, so a late ack cannot
|
||||
promote a lane on the strength of a superseded probe.
|
||||
|
||||
`txAddr[s]` is a single `atomic.Pointer[netip.AddrPort]` that is *both* the TX
|
||||
gate and the destination, so a data-plane routine that loads non-nil has
|
||||
everything it needs in one atomic read and there is no window where one is set
|
||||
and the other is not.
|
||||
|
||||
Probing is driven by the connection manager's per-tunnel traffic tick
|
||||
(`maintainLanes` -> `probeLanes`), which only fires for a tunnel with traffic --
|
||||
the same condition that makes a lane worth having. Timers:
|
||||
|
||||
| timer | value |
|
||||
|---|---|
|
||||
| probe timeout | 2s (shorter than the 5s tick on purpose) |
|
||||
| keepalive | 30s |
|
||||
| retry backoff | 5s, doubling to 60s |
|
||||
| max failure count | 8 |
|
||||
|
||||
Traffic on a lane is not evidence the lane works -- that is the whole reason
|
||||
lanes need probing -- so a lane that silently breaks is only caught by the
|
||||
keepalive.
|
||||
|
||||
Every lane starts out **demanded**, so the first traffic tick probes all of them
|
||||
at once. This matters more than it looks: see "flow pinning" below. A lane that
|
||||
is down also raises demand from the TX path each time a flow wants it, so a
|
||||
lane that keeps failing is retried only while something still wants it, and an
|
||||
idle tunnel costs nothing.
|
||||
|
||||
A lane aims **only** at its own port. There is no fallback to the peer's base
|
||||
port when the lane port doesn't answer: a lane sharing the base port's
|
||||
destination would gain only a source port of its own while costing the peer the
|
||||
receive spread that is the entire point. A lane that can't reach its port stays
|
||||
down and its flows ride the base tunnel.
|
||||
|
||||
## Flow -> lane -> routine: keeping a flow consistent
|
||||
|
||||
This is the part that took the most iterations to get right, so it's worth
|
||||
spelling out why it is shaped the way it is.
|
||||
|
||||
### The kernel's tun queue feedback loop
|
||||
|
||||
On Linux, a multiqueue tun device does not simply hash a flow to a queue. It
|
||||
also *learns*: `tun_flow_update` records "this flow was last seen on queue *q*"
|
||||
from the packets **we write in**, and `tun_automq_select_queue` prefers what it
|
||||
learned over the hash for as long as the flow stays busy.
|
||||
|
||||
We write an inbound packet to the tun queue with the same index as the UDP
|
||||
routine that received it (`batchers[rxc.q]`). So the queue a flow's *outbound*
|
||||
packets arrive on is decided by the socket its *inbound* packets landed on, at
|
||||
the far end, one RTT ago.
|
||||
|
||||
### Why the lane comes from the flow, not the routine
|
||||
|
||||
The obvious design -- routine `q` sends on lane `q` -- deadlocks against that
|
||||
feedback loop:
|
||||
|
||||
1. A tunnel comes up. All lanes are down, so all traffic goes out lane 0.
|
||||
2. The peer receives it all on socket 0 and writes it all to tun queue 0.
|
||||
3. Both kernels now believe every flow belongs on queue 0.
|
||||
4. Every flow is read by routine 0, so every flow picks lane 0.
|
||||
5. Go to 2. The tunnel is pinned to lane 0 for as long as its flows stay busy.
|
||||
|
||||
So the lane comes from **the flow's own 5-tuple**, not from the routine index:
|
||||
|
||||
```go
|
||||
s = (laneFlowHash(fwPacket) + laneBias) % txLanes // lanes.go:txLaneForFlow
|
||||
```
|
||||
|
||||
This makes lane spread completely independent of how the kernel steers tun
|
||||
queues. It also gives the properties you actually want from a flow's point of
|
||||
view:
|
||||
|
||||
- **A flow stays on one lane for its life.** The hash is a pure function of the
|
||||
5-tuple, so there is no per-packet lane hopping and therefore no reordering
|
||||
introduced by multiport.
|
||||
- **One lane's replay window sees one stable set of flows.**
|
||||
- **Both directions of a flow pick partner lanes.** `laneFlowHash` orders the two
|
||||
endpoints before hashing, so it returns the same value from either end, and
|
||||
`laneBias` lines the two sides' choices up. The two directions are exact
|
||||
reverse 4-tuples, which is what NAT and stateful firewalls need.
|
||||
|
||||
`newLaneSet` demanding every lane up front is the other half of this. Lanes have
|
||||
to be up *before* the flows are: a flow that starts while the lanes are still
|
||||
down gets its queue pinned by step 2 above and can stay there for its whole
|
||||
life. Eager demand costs one probe per lane on any tunnel that has traffic, and
|
||||
nothing at all on one that doesn't.
|
||||
|
||||
### The full path
|
||||
|
||||
```
|
||||
outbound inbound (at the peer)
|
||||
-------- ---------------------
|
||||
inside flow
|
||||
| kernel tun hash, or the queue it
|
||||
| learned from our last write
|
||||
v
|
||||
tun queue q -> routine q
|
||||
|
|
||||
| s = laneFlowHash(flow) % txLanes lane s arrives on port base+s
|
||||
v |
|
||||
lane s session | SO_REUSEPORT hash of the
|
||||
| | 4-tuple picks one socket
|
||||
| writers[laneSock(q,s)] v
|
||||
v routine q' (owner of that socket)
|
||||
port base+s -------------------------> |
|
||||
v
|
||||
tun queue q' (teaches the kernel
|
||||
flow -> q')
|
||||
```
|
||||
|
||||
Note that RX is entirely socket-agnostic: the lane comes from the header byte,
|
||||
not from the port the packet arrived on, and roaming is skipped for `lane != 0`
|
||||
(a lane's source address is a per-lane 4-tuple, not the tunnel's remote -- letting
|
||||
it roam the hostinfo would point every non-lane packet at a lane port). So a
|
||||
lane packet may legitimately arrive on any socket, which is what makes the
|
||||
reuseport spread within a port safe.
|
||||
|
||||
### Ordering and locking
|
||||
|
||||
Several routines can write to one socket, since the routines whose lane
|
||||
arithmetic lands on the same index share it. Linux's `batchWriter` serializes
|
||||
`sendmmsg` with a mutex. Per-flow wire order still holds regardless: a flow is
|
||||
hashed onto one lane and read by one routine, so nothing else is writing that
|
||||
flow.
|
||||
|
||||
Each routine holds one `txQueue` with one shared arena (~1.16 MB) and builds a
|
||||
`SendBatch` per lane lazily, on the first packet that picks it, so a routine that
|
||||
never sends on a lane never pays for one.
|
||||
|
||||
### The FIPS 140 encrypt lock
|
||||
|
||||
In FIPS 140 mode -- a `boringcrypto` build, or `GODEBUG=fips140=on` -- nebula
|
||||
uses `noiseutil.CipherAESGCMFIPS140` instead of the plain AES-GCM cipher. That
|
||||
cipher is the TLS 1.3 GCM (`GCMWithXORCounterNonce`), which **panics** if it is
|
||||
asked to seal with a counter that is not strictly greater than the last one. That
|
||||
check is the point: it is the nonce-reuse protection FIPS 140 requires, and
|
||||
`noiseutil`'s startup self-test refuses to run if the check has gone missing.
|
||||
|
||||
The check means encrypts on one session cannot overlap, so
|
||||
`noiseutil.EncryptLockNeeded` is true and every send path takes that session's
|
||||
`ConnectionState.writeLock`:
|
||||
|
||||
```go
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
ci.writeLock.Lock()
|
||||
}
|
||||
c := ci.messageCounter.Add(1)
|
||||
out = header.EncodeLane(scratch, ..., c, lane)
|
||||
out, encErr := ci.eKey.EncryptDanger(out, out, seg, c, nb)
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
ci.writeLock.Unlock()
|
||||
}
|
||||
```
|
||||
|
||||
The lock has to cover the seal and not just the counter increment. Reserving
|
||||
counters atomically is easy; the requirement is that the seals *arrive at the
|
||||
AEAD in counter order*, and only holding the lock across both gives that.
|
||||
|
||||
So under FIPS 140 the encrypt cost of a tunnel is pinned to one core. `routines:
|
||||
N` gives N readers, but every one of them that has a packet for the same peer
|
||||
queues on the same mutex, and with TSO/USO the lock is taken and released once
|
||||
per segment -- up to ~45 times for a single superpacket
|
||||
(`sendInsideEncrypt`). Receive is not affected: `extractFIPSAEAD` deliberately
|
||||
pulls the inner FIPS AEAD out of the `crypto/tls` wrapper because that inner
|
||||
implementation is safe to `Open` concurrently, and `decryptLock` is only ever
|
||||
held around the replay-window check and update.
|
||||
|
||||
Lanes split the lock because a lane is a separate `ConnectionState`: its own
|
||||
`writeLock`, its own message counter, its own nonce sequence. A peer we send to
|
||||
on `txLanes` lanes has `txLanes` independent encrypt locks, so encrypt for that
|
||||
one tunnel can run on that many cores at once.
|
||||
|
||||
The flow hash is what makes this work rather than merely legal. A flow maps to
|
||||
one lane, so all of a flow's packets take one lock in counter order -- exactly
|
||||
what the FIPS AEAD demands -- while different flows to the same peer land on
|
||||
different locks. Contention is divided, not eliminated: any routine may send on
|
||||
any lane, so two routines whose flows hash to the same lane still serialize
|
||||
against each other.
|
||||
|
||||
Two caveats. This is only a FIPS 140 benefit -- in a normal build
|
||||
`EncryptLockNeeded` is false, encrypt is lock-free, and lanes buy path and queue
|
||||
spread only. And it is not unilateral: `txLanes` is bounded by the peer's port
|
||||
count, so a peer that binds one port leaves us back on one encrypt lock however
|
||||
many ports we bound ourselves.
|
||||
|
||||
## Falling back to the old behavior
|
||||
|
||||
Multiport degrades rather than failing, at every level. This is deliberate:
|
||||
managed deployments can't be hard-errored on config they don't control.
|
||||
|
||||
### Whole-node: back to one port
|
||||
|
||||
Any of these turns multiport off, logs why, and leaves a node that binds one
|
||||
port with `routines` `SO_REUSEPORT` sockets -- bit-for-bit the pre-multiport
|
||||
configuration:
|
||||
|
||||
| condition | log |
|
||||
|---|---|
|
||||
| `multiport.enabled: false` | - |
|
||||
| `multiport.ports` unset, 0, or 1 | `multiport disabled: set multiport.ports > 1 ...` |
|
||||
| `listen.port + ports - 1 > 65535` | `multiport disabled: would bind ports beyond 65535` |
|
||||
| platform can't run multiple UDP readers | `multiport disabled: this platform does not support multiple udp readers` |
|
||||
| that capability couldn't be probed | `multiport disabled: could not probe udp reader support` |
|
||||
|
||||
`multiport.ports` is also clamped to 256 (the lane header limit) and to
|
||||
`maxRoutines / routines`, with a warning, rather than being rejected.
|
||||
|
||||
With a dynamic `listen.port: 0`, the first socket binds dynamically and the
|
||||
range is claimed above it; a partially-occupied range re-rolls with a fresh
|
||||
dynamic port up to 6 times.
|
||||
|
||||
`laneSock` collapses to the identity when multiport is off, so the send path is
|
||||
unchanged: a routine writes to its own socket and no lane batches are built at
|
||||
all.
|
||||
|
||||
### Per-tunnel: back to a single session
|
||||
|
||||
`newLaneSet` returns `nil`, and the tunnel is an ordinary one, when:
|
||||
|
||||
- the peer advertised no port count (vanilla peer, or multiport off there), or
|
||||
- `sessions < 2`, i.e. neither side offers a lane.
|
||||
|
||||
`txLanes` can also land at 1 -- we bound ports but the peer bound only one -- in
|
||||
which case the set exists for RX but we never send on a lane.
|
||||
|
||||
### Per-packet: back to the base tunnel
|
||||
|
||||
`txLaneForFlow` returns a nil session and the packet rides the base tunnel,
|
||||
decided per packet with no state to unwind:
|
||||
|
||||
- the flow hashed onto lane 0 (its fair share of flows);
|
||||
- the lane it hashed onto has never come up;
|
||||
- the lane was **demoted** -- a probe or keepalive went unanswered. Fallback is
|
||||
immediate, on the very next packet, and the miss re-raises demand so the lane
|
||||
is re-probed;
|
||||
- the tunnel is **relayed** -- lanes are direct-only, so `probeLanes` drops them
|
||||
all when there is no direct path and rebuilds when one returns;
|
||||
- the peer **roamed** -- a new NAT mapping has no derivable relationship to the
|
||||
old lane ports, so every lane is torn down and re-probed from a clean backoff.
|
||||
Standing demand is deliberately kept across a reset, so the lanes that were
|
||||
actually carrying data come back first.
|
||||
|
||||
### Always on the base tunnel
|
||||
|
||||
Handshakes, lighthouse traffic, punching, relay carriers, close packets, rejects,
|
||||
and lane probe *acks* all use the base session and a socket on the base port
|
||||
(`egressSock`). Base traffic must keep the base source port or a vanilla peer
|
||||
would see the tunnel's address move and roam-thrash.
|
||||
|
||||
`recv_error` is the one deliberate exception: it replies from the socket the
|
||||
offending packet arrived on, because a lane peer's spoof guard compares our
|
||||
source address against that lane's remote and would discard a reply from the
|
||||
base port.
|
||||
|
||||
### Wire compatibility
|
||||
|
||||
- A vanilla sender emits lane 0 in a field it thinks is reserved, and lane 0 is
|
||||
the base tunnel, so it reads correctly with no version check.
|
||||
- We always send the upper 8 reserved bits as zero.
|
||||
- A lane index above what this tunnel has (a stale lane from a rolled tunnel, or
|
||||
a peer sending above what it advertised) is dropped **silently** -- a
|
||||
`recv_error` would tear down a perfectly good base tunnel on the strength of
|
||||
one odd packet.
|
||||
- Lane ciphertext wrapped in a relay carrier is refused before the session
|
||||
lookup, so a junk relay packet can't make us derive a session.
|
||||
|
||||
## Security notes
|
||||
|
||||
- The lane index is in the AEAD's associated data, so it is authenticated, not
|
||||
just carried.
|
||||
- On RX, a lane session derived for an unrecognized lane is **not installed**
|
||||
until the packet actually decrypts. Anyone who can spoof this tunnel's local
|
||||
index can name any lane; installing on sight would let them make us hold a
|
||||
replay window and two cipher states per lane, per tunnel, for lanes they never
|
||||
send on.
|
||||
- Two routines racing on a lane's first packet each derive a session. The loser's
|
||||
is dropped, and its replay-window entry for the packet it just accepted is
|
||||
handed to the winner (`installSession`) -- the keys are identical, so it is the
|
||||
same window in every respect that matters.
|
||||
|
||||
## Configuration
|
||||
|
||||
```yaml
|
||||
routines: 8 # PER PORT under multiport; total workers = routines * ports
|
||||
multiport:
|
||||
enabled: true # default true, but inert without ports
|
||||
ports: 4 # consecutive ports from listen.port; must be > 1, no default
|
||||
lanes: 0 # 0 = one per bound port; lower to send on a subset
|
||||
```
|
||||
|
||||
`multiport.ports` has no default on purpose: under these semantics a default
|
||||
would silently multiply the worker count. Nothing here is reloadable.
|
||||
|
||||
Both sides need a port range, and the range
|
||||
`[listen.port, listen.port+ports-1]` must be open in both directions. Opening
|
||||
only the base port is the common failure and gives you exactly one working lane
|
||||
-- the one whose rotation happens to land on the base port.
|
||||
|
||||
## Observability
|
||||
|
||||
```
|
||||
nebula-ssh> print-tunnel -vpn-addr <peer>
|
||||
```
|
||||
|
||||
`lanes[]` gives per-lane `up`, `remote` and `messageCounter`, which is the fastest
|
||||
way to tell "no lanes negotiated" (key absent) from "lanes up but traffic on one"
|
||||
(counters).
|
||||
|
||||
Metrics, registered only when multiport is running so they don't sit at zero on
|
||||
nodes without it:
|
||||
|
||||
- `multiport.lanes.up` -- lanes currently carrying traffic
|
||||
- `multiport.lanes.tunnels` -- tunnels with any lanes
|
||||
|
||||
Both are counted by walking the hostmap rather than kept at promotion/demotion,
|
||||
because a counter would drift upward forever: a tunnel torn down with its lanes
|
||||
up never demotes them.
|
||||
|
||||
Logs worth grepping: `multiport enabled` and `multiport routines` at startup, the
|
||||
`lanes` attr on handshake completion (`tx`, `sessions`, `peerBasePort`,
|
||||
`peerPorts`, `portOffset`), and `Multiport lane up` / `Multiport lane demoted`,
|
||||
both of which name the `udpAddr` involved.
|
||||
|
||||
## Known gaps
|
||||
|
||||
- No `readOutsidePackets`-level test for the RX lane drop paths.
|
||||
- No multiport coverage in the e2e suite.
|
||||
- `routines * ports > 256` fails at startup from the kernel's tun queue limit
|
||||
rather than being clamped with a warning.
|
||||
@@ -179,9 +179,54 @@ listen:
|
||||
# Currently, this defaults to 1 which means we have 1 tun queue reader and 1
|
||||
# UDP queue reader. Setting this above one will set IFF_MULTI_QUEUE on the tun
|
||||
# device and SO_REUSEPORT on the UDP socket to allow multiple queues.
|
||||
# With multiport enabled this is the number of routines *per port*, so the total
|
||||
# is routines * multiport.ports.
|
||||
# This option is only supported on Linux.
|
||||
#routines: 1
|
||||
|
||||
# EXPERIMENTAL: multiport lanes give each pair of hosts multiple underlay UDP
|
||||
# flows so overlay traffic is no longer bottlenecked by a single 5-tuple
|
||||
# (one ECMP path, one NIC RSS queue, one per-flow policer). Instead of every
|
||||
# socket sharing listen.port, multiport.ports consecutive ports are bound and
|
||||
# each gets its own group of `routines` sockets sharing it via SO_REUSEPORT, so
|
||||
# no port (the base port above all, which carries every handshake and every
|
||||
# vanilla peer) depends on a single core. One extra tunnel ("lane") per port is
|
||||
# negotiated with capable peers: lane i handshakes from local port listen.port+i
|
||||
# to the peer's advertised base+((i + pair_offset) mod peer_ports), where
|
||||
# pair_offset is a per-pair hash that spreads many small peers across a big
|
||||
# peer's whole port range.
|
||||
# Each lane is a full Noise session with its own
|
||||
# keys, nonce counter and replay window, so flows taking different paths
|
||||
# never fight over shared replay state.
|
||||
#
|
||||
# Peers negotiate lanes in the handshake; vanilla peers get a single normal
|
||||
# tunnel. All control traffic (handshakes, lighthouse, punching, relays) and
|
||||
# the data fallback stay on the base tunnel/port. Lanes are built lazily: a
|
||||
# lane is only established once a routine actually has traffic for that peer
|
||||
# and no lane to carry it, so a peer you exchange a trickle with costs one
|
||||
# tunnel regardless of how many routines are configured. Established lanes
|
||||
# are kept alive with their own keepalives, and traffic falls back to the base
|
||||
# tunnel while a lane is down or not yet up.
|
||||
#
|
||||
# Requirements: multiport.ports > 1, Linux, and the port range
|
||||
# [listen.port, listen.port+multiport.ports-1] reachable through firewalls on
|
||||
# both sides. A lane whose port is unreachable stays down and its traffic rides
|
||||
# the base tunnel. With listen.port 0 the base port is dynamic and the next
|
||||
# ports-1 ports above it are claimed. Degrades to a single port when the
|
||||
# requirements don't hold. Not reloadable.
|
||||
#multiport:
|
||||
#enabled: true
|
||||
# How many consecutive UDP ports to bind, starting at listen.port. Must be
|
||||
# greater than 1 for multiport to do anything; there is no default, since
|
||||
# each port costs a full set of `routines` threads and sockets. Capped at 256
|
||||
# (the lane header limit). 4-8 is plenty to escape a single ECMP path.
|
||||
#ports: 0
|
||||
# How many lanes to run, counting the base tunnel as lane 0. 0 (default)
|
||||
# means one per bound port. Lowering this sends on a subset of the range,
|
||||
# which bounds how many extra tunnels each peer pair maintains (useful on a
|
||||
# big server with many peers); the ports are bound and read either way.
|
||||
#lanes: 0
|
||||
|
||||
punchy:
|
||||
# Continues to punch inbound/outbound at a regular interval to avoid expiration of firewall nat mappings
|
||||
# This setting is reloadable.
|
||||
|
||||
@@ -26,4 +26,18 @@ message NebulaHandshakeDetails {
|
||||
uint32 CertVersion = 8;
|
||||
// reserved for WIP multiport
|
||||
reserved 6, 7;
|
||||
// Multiport lane negotiation. Absent on hosts without multiport enabled;
|
||||
// vanilla nebula treats 9 and 10 as unknown fields and skips them.
|
||||
LaneDetails InitiatorLanes = 9;
|
||||
LaneDetails ResponderLanes = 10;
|
||||
}
|
||||
|
||||
// LaneDetails advertises a host's multiport lane capability. On a base
|
||||
// handshake LaneIndex is 0 and PortCount/BasePort describe the sender's
|
||||
// consecutively bound UDP ports. On a lane handshake the initiator sets
|
||||
// LaneIndex to its (nonzero) lane number.
|
||||
message LaneDetails {
|
||||
uint32 PortCount = 1;
|
||||
uint32 BasePort = 2;
|
||||
uint32 LaneIndex = 3;
|
||||
}
|
||||
|
||||
@@ -71,6 +71,7 @@ func newTestMachine(
|
||||
cs.version, cs.getCredential,
|
||||
verifier, func() (uint32, error) { return localIndex, nil },
|
||||
initiator, header.HandshakeIXPSK0,
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
return m
|
||||
|
||||
+34
-1
@@ -39,6 +39,14 @@ type Result struct {
|
||||
HandshakeTime uint64
|
||||
MessageIndex uint64 // number of messages exchanged during the handshake
|
||||
Initiator bool
|
||||
|
||||
// Multiport lane negotiation, from the peer's LaneDetails. All zero when
|
||||
// the peer did not advertise (vanilla peer or multiport disabled).
|
||||
// PeerTxLanes is how many lanes the peer may send on, which is how many lane
|
||||
// sessions we need in order to receive everything it sends.
|
||||
PeerPortCount uint32
|
||||
PeerBasePort uint32
|
||||
PeerTxLanes uint32
|
||||
}
|
||||
|
||||
// Machine drives a Noise handshake through N messages. It handles Noise
|
||||
@@ -61,6 +69,7 @@ type Machine struct {
|
||||
verifier CertVerifier
|
||||
result *Result
|
||||
msgs []msgFlags
|
||||
lanes *LaneDetails // our multiport advert; nil emits a vanilla payload
|
||||
myVersion cert.Version
|
||||
subtype header.MessageSubType
|
||||
indexAllocated bool
|
||||
@@ -73,6 +82,8 @@ type Machine struct {
|
||||
// the noise pattern and the per-message content layout. The credential for
|
||||
// `version` is fetched via getCred and used to seed the noise.HandshakeState.
|
||||
// IndexAllocator is called lazily when the first outgoing payload is built.
|
||||
// lanes, when non-nil, is emitted as this side's multiport advert on every
|
||||
// payload-bearing message; nil produces byte-identical vanilla payloads.
|
||||
func NewMachine(
|
||||
version cert.Version,
|
||||
getCred GetCredentialFunc,
|
||||
@@ -80,6 +91,7 @@ func NewMachine(
|
||||
allocIndex IndexAllocator,
|
||||
initiator bool,
|
||||
subtype header.MessageSubType,
|
||||
lanes *LaneDetails,
|
||||
) (*Machine, error) {
|
||||
info, err := subtypeInfoFor(subtype)
|
||||
if err != nil {
|
||||
@@ -103,6 +115,7 @@ func NewMachine(
|
||||
getCred: getCred,
|
||||
allocIndex: allocIndex,
|
||||
verifier: verifier,
|
||||
lanes: lanes,
|
||||
myVersion: version,
|
||||
result: &Result{
|
||||
Initiator: initiator,
|
||||
@@ -298,7 +311,8 @@ func (m *Machine) processPayload(msg []byte, flags msgFlags) error {
|
||||
}
|
||||
|
||||
// Assert the payload contains exactly what we expect
|
||||
hasPayloadData := payload.InitiatorIndex != 0 || payload.ResponderIndex != 0 || payload.Time != 0
|
||||
hasPayloadData := payload.InitiatorIndex != 0 || payload.ResponderIndex != 0 || payload.Time != 0 ||
|
||||
payload.InitiatorLanes != nil || payload.ResponderLanes != nil
|
||||
if hasPayloadData != flags.expectsPayload {
|
||||
m.failed = true
|
||||
return ErrUnexpectedContent
|
||||
@@ -327,6 +341,23 @@ func (m *Machine) processPayload(msg []byte, flags msgFlags) error {
|
||||
m.result.RemoteIndex = remoteIndex
|
||||
m.result.HandshakeTime = payload.Time
|
||||
m.payloadSet = true
|
||||
|
||||
// Multiport advert from the peer's side of the exchange. Out-of-range
|
||||
// values mean a peer we can't pair lanes with; ignore the advert
|
||||
// rather than failing the handshake — the tunnel itself is fine, it
|
||||
// just won't get lanes. Semantic policing (port-count caps, lane
|
||||
// clamping) belongs to the handshake manager.
|
||||
var peerLanes *LaneDetails
|
||||
if m.result.Initiator {
|
||||
peerLanes = payload.ResponderLanes
|
||||
} else {
|
||||
peerLanes = payload.InitiatorLanes
|
||||
}
|
||||
if peerLanes != nil && peerLanes.BasePort <= 0xffff && peerLanes.PortCount <= 0xffff && peerLanes.TxLanes <= 0xffff {
|
||||
m.result.PeerPortCount = peerLanes.PortCount
|
||||
m.result.PeerBasePort = peerLanes.BasePort
|
||||
m.result.PeerTxLanes = peerLanes.TxLanes
|
||||
}
|
||||
}
|
||||
|
||||
// Process certificate
|
||||
@@ -397,9 +428,11 @@ func (m *Machine) marshalOutgoing(flags msgFlags) ([]byte, error) {
|
||||
|
||||
if m.result.Initiator {
|
||||
p.InitiatorIndex = m.result.LocalIndex
|
||||
p.InitiatorLanes = m.lanes
|
||||
} else {
|
||||
p.ResponderIndex = m.result.LocalIndex
|
||||
p.InitiatorIndex = m.result.RemoteIndex
|
||||
p.ResponderLanes = m.lanes
|
||||
}
|
||||
p.Time = uint64(time.Now().UnixNano())
|
||||
}
|
||||
|
||||
@@ -0,0 +1,111 @@
|
||||
package handshake
|
||||
|
||||
import (
|
||||
"net/netip"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/slackhq/nebula/cert"
|
||||
ct "github.com/slackhq/nebula/cert_test"
|
||||
"github.com/slackhq/nebula/header"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// newTestLaneMachine is newTestMachine with a lane advert attached.
|
||||
func newTestLaneMachine(
|
||||
t *testing.T,
|
||||
cs *testCertState,
|
||||
verifier CertVerifier,
|
||||
initiator bool,
|
||||
localIndex uint32,
|
||||
lanes *LaneDetails,
|
||||
) *Machine {
|
||||
t.Helper()
|
||||
m, err := NewMachine(
|
||||
cs.version, cs.getCredential,
|
||||
verifier, func() (uint32, error) { return localIndex, nil },
|
||||
initiator, header.HandshakeIXPSK0,
|
||||
lanes,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
return m
|
||||
}
|
||||
|
||||
func doFullLaneHandshake(t *testing.T, initLanes, respLanes *LaneDetails) (initR, respR *Result) {
|
||||
t.Helper()
|
||||
ca, _, caKey, _ := ct.NewTestCaCert(
|
||||
cert.Version2, cert.Curve_CURVE25519, time.Time{}, time.Time{}, nil, nil, nil,
|
||||
)
|
||||
caPool := ct.NewTestCAPool(ca)
|
||||
v := testVerifier(caPool)
|
||||
|
||||
initCS := newTestCertState(t, ca, caKey, "initiator", []netip.Prefix{netip.MustParsePrefix("10.0.0.1/24")})
|
||||
respCS := newTestCertState(t, ca, caKey, "responder", []netip.Prefix{netip.MustParsePrefix("10.0.0.2/24")})
|
||||
|
||||
initM := newTestLaneMachine(t, initCS, v, true, 1000, initLanes)
|
||||
respM := newTestLaneMachine(t, respCS, v, false, 2000, respLanes)
|
||||
|
||||
msg1, err := initM.Initiate(nil)
|
||||
require.NoError(t, err)
|
||||
|
||||
resp, respR, err := respM.ProcessPacket(nil, msg1)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, respR)
|
||||
|
||||
_, initR, err = initM.ProcessPacket(nil, resp)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, initR)
|
||||
return initR, respR
|
||||
}
|
||||
|
||||
func TestMachineLaneAdvertBothSides(t *testing.T) {
|
||||
initR, respR := doFullLaneHandshake(t,
|
||||
&LaneDetails{PortCount: 8, BasePort: 4242, TxLanes: 8},
|
||||
&LaneDetails{PortCount: 4, BasePort: 5353, TxLanes: 3},
|
||||
)
|
||||
|
||||
// Each side's Result carries the peer's advert.
|
||||
assert.Equal(t, uint32(4), initR.PeerPortCount)
|
||||
assert.Equal(t, uint32(5353), initR.PeerBasePort)
|
||||
assert.Equal(t, uint32(3), initR.PeerTxLanes)
|
||||
|
||||
assert.Equal(t, uint32(8), respR.PeerPortCount)
|
||||
assert.Equal(t, uint32(4242), respR.PeerBasePort)
|
||||
assert.Equal(t, uint32(8), respR.PeerTxLanes)
|
||||
}
|
||||
|
||||
func TestMachineLaneAdvertAsymmetric(t *testing.T) {
|
||||
// Vanilla initiator, multiport responder and vice versa: the nil side
|
||||
// yields all-zero peer fields on the other end.
|
||||
initR, respR := doFullLaneHandshake(t, nil, &LaneDetails{PortCount: 4, BasePort: 5353})
|
||||
assert.Equal(t, uint32(4), initR.PeerPortCount)
|
||||
assert.Equal(t, uint32(0), respR.PeerPortCount)
|
||||
assert.Equal(t, uint32(0), respR.PeerBasePort)
|
||||
|
||||
initR, respR = doFullLaneHandshake(t, &LaneDetails{PortCount: 8, BasePort: 4242}, nil)
|
||||
assert.Equal(t, uint32(0), initR.PeerPortCount)
|
||||
assert.Equal(t, uint32(8), respR.PeerPortCount)
|
||||
}
|
||||
|
||||
func TestMachineLaneAdvertOutOfRangeIgnored(t *testing.T) {
|
||||
// A BasePort that can't be a real UDP port is ignored, not fatal.
|
||||
initR, respR := doFullLaneHandshake(t,
|
||||
&LaneDetails{PortCount: 8, BasePort: 70000, TxLanes: 8},
|
||||
&LaneDetails{PortCount: 4, BasePort: 5353, TxLanes: 4},
|
||||
)
|
||||
assert.Equal(t, uint32(0), respR.PeerPortCount)
|
||||
assert.Equal(t, uint32(0), respR.PeerBasePort)
|
||||
assert.Equal(t, uint32(0), respR.PeerTxLanes)
|
||||
// The sane side still negotiates.
|
||||
assert.Equal(t, uint32(4), initR.PeerPortCount)
|
||||
|
||||
// An out-of-range TxLanes drops the whole advert the same way: a lane index
|
||||
// that does not fit the header is as unusable as an impossible port.
|
||||
initR, respR = doFullLaneHandshake(t,
|
||||
&LaneDetails{PortCount: 8, BasePort: 4242, TxLanes: 0x10000},
|
||||
&LaneDetails{PortCount: 4, BasePort: 5353, TxLanes: 4},
|
||||
)
|
||||
assert.Equal(t, uint32(0), respR.PeerPortCount)
|
||||
assert.Equal(t, uint32(4), initR.PeerPortCount)
|
||||
}
|
||||
@@ -444,6 +444,7 @@ func TestMachineThreeMessagePattern(t *testing.T) {
|
||||
initCS.getCredential, v,
|
||||
func() (uint32, error) { return 1000, nil },
|
||||
true, header.HandshakeXXPSK0,
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
|
||||
@@ -452,6 +453,7 @@ func TestMachineThreeMessagePattern(t *testing.T) {
|
||||
respCS.getCredential, v,
|
||||
func() (uint32, error) { return 2000, nil },
|
||||
false, header.HandshakeXXPSK0,
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
|
||||
|
||||
+120
-5
@@ -20,15 +20,40 @@ type Payload struct {
|
||||
ResponderIndex uint32
|
||||
Time uint64
|
||||
CertVersion uint32
|
||||
|
||||
// Multiport lane negotiation; nil when the sender has multiport disabled
|
||||
// (which keeps the encoded payload byte-identical to a vanilla one).
|
||||
InitiatorLanes *LaneDetails
|
||||
ResponderLanes *LaneDetails
|
||||
}
|
||||
|
||||
// LaneDetails advertises multiport lane capability: the contiguous UDP port
|
||||
// range the sender bound, and how many lanes it may send on. The receiver needs
|
||||
// PortCount/BasePort to aim its own lanes and TxLanes to know how many lane
|
||||
// sessions to derive for receiving.
|
||||
type LaneDetails struct {
|
||||
PortCount uint32
|
||||
BasePort uint32
|
||||
TxLanes uint32
|
||||
}
|
||||
|
||||
// Proto field numbers for NebulaHandshakeDetails
|
||||
const (
|
||||
fieldCert = 1 // bytes
|
||||
fieldInitiatorIndex = 2 // uint32
|
||||
fieldResponderIndex = 3 // uint32
|
||||
fieldTime = 5 // uint64
|
||||
fieldCertVersion = 8 // uint32
|
||||
fieldCert = 1 // bytes
|
||||
fieldInitiatorIndex = 2 // uint32
|
||||
fieldResponderIndex = 3 // uint32
|
||||
fieldTime = 5 // uint64
|
||||
fieldCertVersion = 8 // uint32
|
||||
fieldInitiatorLanes = 9 // LaneDetails
|
||||
fieldResponderLanes = 10 // LaneDetails
|
||||
)
|
||||
|
||||
// Proto field numbers for LaneDetails.
|
||||
// Field 3 was a per-lane handshake index and is permanently reserved.
|
||||
const (
|
||||
fieldLanePortCount = 1 // uint32
|
||||
fieldLaneBasePort = 2 // uint32
|
||||
fieldLaneTxLanes = 4 // uint32
|
||||
)
|
||||
|
||||
// MarshalPayload encodes a handshake payload in protobuf wire format compatible
|
||||
@@ -57,6 +82,16 @@ func MarshalPayload(out []byte, p Payload) []byte {
|
||||
details = protowire.AppendTag(details, fieldCertVersion, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, uint64(p.CertVersion))
|
||||
}
|
||||
// Emitted last to keep the encoding in ascending field-number order, which
|
||||
// is what protoc-gen-go would produce for the same message.
|
||||
if p.InitiatorLanes != nil {
|
||||
details = protowire.AppendTag(details, fieldInitiatorLanes, protowire.BytesType)
|
||||
details = protowire.AppendBytes(details, p.InitiatorLanes.marshal(nil))
|
||||
}
|
||||
if p.ResponderLanes != nil {
|
||||
details = protowire.AppendTag(details, fieldResponderLanes, protowire.BytesType)
|
||||
details = protowire.AppendBytes(details, p.ResponderLanes.marshal(nil))
|
||||
}
|
||||
|
||||
out = protowire.AppendTag(out, 1, protowire.BytesType)
|
||||
out = protowire.AppendBytes(out, details)
|
||||
@@ -64,6 +99,20 @@ func MarshalPayload(out []byte, p Payload) []byte {
|
||||
return out
|
||||
}
|
||||
|
||||
// marshal appends the LaneDetails submessage fields to out. All fields are
|
||||
// emitted unconditionally: a LaneDetails is only present at all when multiport
|
||||
// is negotiating, and explicit zeros keep the parser's presence semantics
|
||||
// trivial.
|
||||
func (d *LaneDetails) marshal(out []byte) []byte {
|
||||
out = protowire.AppendTag(out, fieldLanePortCount, protowire.VarintType)
|
||||
out = protowire.AppendVarint(out, uint64(d.PortCount))
|
||||
out = protowire.AppendTag(out, fieldLaneBasePort, protowire.VarintType)
|
||||
out = protowire.AppendVarint(out, uint64(d.BasePort))
|
||||
out = protowire.AppendTag(out, fieldLaneTxLanes, protowire.VarintType)
|
||||
out = protowire.AppendVarint(out, uint64(d.TxLanes))
|
||||
return out
|
||||
}
|
||||
|
||||
// UnmarshalPayload decodes a protobuf-encoded NebulaHandshake message.
|
||||
func UnmarshalPayload(b []byte) (Payload, error) {
|
||||
var p Payload
|
||||
@@ -161,6 +210,72 @@ func unmarshalPayloadDetails(p *Payload, b []byte) error {
|
||||
}
|
||||
p.CertVersion = uint32(v)
|
||||
b = b[n:]
|
||||
case fieldInitiatorLanes:
|
||||
if typ != protowire.BytesType {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
v, n := protowire.ConsumeBytes(b)
|
||||
if n < 0 {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
p.InitiatorLanes = new(LaneDetails)
|
||||
if err := unmarshalLaneDetails(p.InitiatorLanes, v); err != nil {
|
||||
return err
|
||||
}
|
||||
b = b[n:]
|
||||
case fieldResponderLanes:
|
||||
if typ != protowire.BytesType {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
v, n := protowire.ConsumeBytes(b)
|
||||
if n < 0 {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
p.ResponderLanes = new(LaneDetails)
|
||||
if err := unmarshalLaneDetails(p.ResponderLanes, v); err != nil {
|
||||
return err
|
||||
}
|
||||
b = b[n:]
|
||||
default:
|
||||
n := protowire.ConsumeFieldValue(num, typ, b)
|
||||
if n < 0 {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
b = b[n:]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func unmarshalLaneDetails(d *LaneDetails, b []byte) error {
|
||||
for len(b) > 0 {
|
||||
num, typ, n := protowire.ConsumeTag(b)
|
||||
if n < 0 {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
b = b[n:]
|
||||
|
||||
// Same contract as the details parser: known fields hard-fail on a
|
||||
// wire-type mismatch, unknown fields are skipped, repeated singular
|
||||
// fields follow proto3 last-wins.
|
||||
switch num {
|
||||
case fieldLanePortCount, fieldLaneBasePort, fieldLaneTxLanes:
|
||||
if typ != protowire.VarintType {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
v, n := protowire.ConsumeVarint(b)
|
||||
if n < 0 || v > math.MaxUint32 {
|
||||
return errInvalidHandshakeDetails
|
||||
}
|
||||
switch num {
|
||||
case fieldLanePortCount:
|
||||
d.PortCount = uint32(v)
|
||||
case fieldLaneBasePort:
|
||||
d.BasePort = uint32(v)
|
||||
case fieldLaneTxLanes:
|
||||
d.TxLanes = uint32(v)
|
||||
}
|
||||
b = b[n:]
|
||||
default:
|
||||
n := protowire.ConsumeFieldValue(num, typ, b)
|
||||
if n < 0 {
|
||||
|
||||
+137
-11
@@ -117,23 +117,134 @@ func TestPayloadUnknownFields(t *testing.T) {
|
||||
assert.Equal(t, uint32(88), got.ResponderIndex)
|
||||
})
|
||||
|
||||
t.Run("reserved fields 6 and 7 are skipped", func(t *testing.T) {
|
||||
// Fields 6 and 7 are reserved in the proto definition
|
||||
t.Run("unknown field inside LaneDetails is skipped", func(t *testing.T) {
|
||||
var lane []byte
|
||||
lane = protowire.AppendTag(lane, fieldLanePortCount, protowire.VarintType)
|
||||
lane = protowire.AppendVarint(lane, 4)
|
||||
lane = protowire.AppendTag(lane, 50, protowire.VarintType) // unknown subfield
|
||||
lane = protowire.AppendVarint(lane, 9999)
|
||||
lane = protowire.AppendTag(lane, fieldLaneBasePort, protowire.VarintType)
|
||||
lane = protowire.AppendVarint(lane, 4242)
|
||||
|
||||
var details []byte
|
||||
details = protowire.AppendTag(details, fieldInitiatorIndex, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, 100)
|
||||
details = protowire.AppendTag(details, 6, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, 1)
|
||||
details = protowire.AppendTag(details, 7, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, 2)
|
||||
details = protowire.AppendTag(details, fieldInitiatorLanes, protowire.BytesType)
|
||||
details = protowire.AppendBytes(details, lane)
|
||||
|
||||
var data []byte
|
||||
data = protowire.AppendTag(data, 1, protowire.BytesType)
|
||||
data = protowire.AppendBytes(data, details)
|
||||
got, err := UnmarshalPayload(wrapDetails(details))
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint32(100), got.InitiatorIndex)
|
||||
require.NotNil(t, got.InitiatorLanes)
|
||||
assert.Equal(t, uint32(4), got.InitiatorLanes.PortCount)
|
||||
assert.Equal(t, uint32(4242), got.InitiatorLanes.BasePort)
|
||||
})
|
||||
}
|
||||
|
||||
func TestPayloadLaneDetails(t *testing.T) {
|
||||
t.Run("round trip both sides", func(t *testing.T) {
|
||||
data := MarshalPayload(nil, Payload{
|
||||
InitiatorIndex: 12345,
|
||||
Time: 999,
|
||||
InitiatorLanes: &LaneDetails{PortCount: 8, BasePort: 4242, TxLanes: 3},
|
||||
ResponderLanes: &LaneDetails{PortCount: 4, BasePort: 5353},
|
||||
})
|
||||
|
||||
got, err := UnmarshalPayload(data)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint32(100), got.InitiatorIndex)
|
||||
require.NotNil(t, got.InitiatorLanes)
|
||||
assert.Equal(t, LaneDetails{PortCount: 8, BasePort: 4242, TxLanes: 3}, *got.InitiatorLanes)
|
||||
require.NotNil(t, got.ResponderLanes)
|
||||
assert.Equal(t, LaneDetails{PortCount: 4, BasePort: 5353}, *got.ResponderLanes)
|
||||
})
|
||||
|
||||
t.Run("zero-valued LaneDetails survives the round trip", func(t *testing.T) {
|
||||
// Presence is what negotiation keys on; an all-zero advert must not
|
||||
// decay to nil.
|
||||
data := MarshalPayload(nil, Payload{
|
||||
InitiatorIndex: 1,
|
||||
InitiatorLanes: &LaneDetails{},
|
||||
})
|
||||
got, err := UnmarshalPayload(data)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, got.InitiatorLanes)
|
||||
assert.Equal(t, LaneDetails{}, *got.InitiatorLanes)
|
||||
assert.Nil(t, got.ResponderLanes)
|
||||
})
|
||||
|
||||
t.Run("nil lanes marshal byte-identical to a vanilla payload", func(t *testing.T) {
|
||||
p := Payload{
|
||||
Cert: []byte("cert"),
|
||||
CertVersion: 2,
|
||||
InitiatorIndex: 100,
|
||||
Time: 999,
|
||||
}
|
||||
// The vanilla encoding of the same fields, built by hand in field order.
|
||||
var details []byte
|
||||
details = protowire.AppendTag(details, fieldCert, protowire.BytesType)
|
||||
details = protowire.AppendBytes(details, p.Cert)
|
||||
details = protowire.AppendTag(details, fieldInitiatorIndex, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, uint64(p.InitiatorIndex))
|
||||
details = protowire.AppendTag(details, fieldTime, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, p.Time)
|
||||
details = protowire.AppendTag(details, fieldCertVersion, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, uint64(p.CertVersion))
|
||||
|
||||
assert.Equal(t, wrapDetails(details), MarshalPayload(nil, p))
|
||||
})
|
||||
|
||||
t.Run("lane field with wrong wire type rejected", func(t *testing.T) {
|
||||
for _, field := range []protowire.Number{fieldInitiatorLanes, fieldResponderLanes} {
|
||||
var details []byte
|
||||
details = protowire.AppendTag(details, field, protowire.VarintType)
|
||||
details = protowire.AppendVarint(details, 1)
|
||||
_, err := UnmarshalPayload(wrapDetails(details))
|
||||
assert.Error(t, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("lane subfield with wrong wire type rejected", func(t *testing.T) {
|
||||
var lane []byte
|
||||
lane = protowire.AppendTag(lane, fieldLanePortCount, protowire.BytesType)
|
||||
lane = protowire.AppendBytes(lane, []byte{1, 2, 3})
|
||||
|
||||
var details []byte
|
||||
details = protowire.AppendTag(details, fieldInitiatorLanes, protowire.BytesType)
|
||||
details = protowire.AppendBytes(details, lane)
|
||||
_, err := UnmarshalPayload(wrapDetails(details))
|
||||
assert.Error(t, err)
|
||||
})
|
||||
|
||||
t.Run("truncated LaneDetails submessage rejected", func(t *testing.T) {
|
||||
var details []byte
|
||||
details = protowire.AppendTag(details, fieldInitiatorLanes, protowire.BytesType)
|
||||
details = append(details, 0x0a, 0x01, 0x02) // length 10, only 2 bytes
|
||||
_, err := UnmarshalPayload(wrapDetails(details))
|
||||
assert.Error(t, err)
|
||||
})
|
||||
|
||||
t.Run("truncated varint inside LaneDetails rejected", func(t *testing.T) {
|
||||
var lane []byte
|
||||
lane = protowire.AppendTag(lane, fieldLaneBasePort, protowire.VarintType)
|
||||
lane = append(lane, 0x80) // incomplete varint
|
||||
|
||||
var details []byte
|
||||
details = protowire.AppendTag(details, fieldResponderLanes, protowire.BytesType)
|
||||
details = protowire.AppendBytes(details, lane)
|
||||
_, err := UnmarshalPayload(wrapDetails(details))
|
||||
assert.Error(t, err)
|
||||
})
|
||||
|
||||
t.Run("lane subfield varint overflow rejected", func(t *testing.T) {
|
||||
var lane []byte
|
||||
lane = protowire.AppendTag(lane, fieldLaneTxLanes, protowire.VarintType)
|
||||
lane = protowire.AppendVarint(lane, math.MaxUint32+1)
|
||||
|
||||
var details []byte
|
||||
details = protowire.AppendTag(details, fieldInitiatorLanes, protowire.BytesType)
|
||||
details = protowire.AppendBytes(details, lane)
|
||||
_, err := UnmarshalPayload(wrapDetails(details))
|
||||
assert.Error(t, err)
|
||||
})
|
||||
}
|
||||
|
||||
@@ -328,6 +439,12 @@ func FuzzPayload(f *testing.F) {
|
||||
Time: 3,
|
||||
CertVersion: 2,
|
||||
}))
|
||||
f.Add(MarshalPayload(nil, Payload{
|
||||
InitiatorIndex: 1,
|
||||
Time: 3,
|
||||
InitiatorLanes: &LaneDetails{PortCount: 8, BasePort: 4242, TxLanes: 2},
|
||||
ResponderLanes: &LaneDetails{PortCount: 4, BasePort: 5353},
|
||||
}))
|
||||
f.Add([]byte{})
|
||||
f.Add([]byte{0xff})
|
||||
|
||||
@@ -357,5 +474,14 @@ func payloadsEqual(a, b Payload) bool {
|
||||
a.InitiatorIndex == b.InitiatorIndex &&
|
||||
a.ResponderIndex == b.ResponderIndex &&
|
||||
a.Time == b.Time &&
|
||||
a.CertVersion == b.CertVersion
|
||||
a.CertVersion == b.CertVersion &&
|
||||
laneDetailsEqual(a.InitiatorLanes, b.InitiatorLanes) &&
|
||||
laneDetailsEqual(a.ResponderLanes, b.ResponderLanes)
|
||||
}
|
||||
|
||||
func laneDetailsEqual(a, b *LaneDetails) bool {
|
||||
if a == nil || b == nil {
|
||||
return a == b
|
||||
}
|
||||
return *a == *b
|
||||
}
|
||||
|
||||
+60
-4
@@ -50,6 +50,14 @@ type HandshakeConfig struct {
|
||||
retries int64
|
||||
triggerBuffer int
|
||||
|
||||
// Multiport lane parameters; laneCount == 0 means multiport is disabled.
|
||||
// laneCount includes implicit lane 0 (the base tunnel), so lanes
|
||||
// 1..laneCount-1 may carry traffic. lanePortCount/laneBasePort describe our
|
||||
// own bound port range and are advertised in every handshake payload.
|
||||
laneCount int
|
||||
lanePortCount uint16
|
||||
laneBasePort uint16
|
||||
|
||||
messageMetrics *MessageMetrics
|
||||
}
|
||||
|
||||
@@ -535,6 +543,9 @@ func (hm *HandshakeManager) DeleteHostInfo(hostinfo *HostInfo) {
|
||||
|
||||
func (hm *HandshakeManager) unlockedDeleteHostInfo(hostinfo *HostInfo) {
|
||||
for _, addr := range hostinfo.vpnAddrs {
|
||||
// Only delete the pending entry if it is actually ours: an
|
||||
// unconditional delete could evict a concurrently pending handshake for
|
||||
// the same address.
|
||||
if cur, ok := hm.vpnIps[addr]; ok && cur.hostinfo == hostinfo {
|
||||
delete(hm.vpnIps, addr)
|
||||
}
|
||||
@@ -672,6 +683,7 @@ func (hm *HandshakeManager) buildStage0Packet(hh *HandshakeHostInfo) bool {
|
||||
v, cs.GetCredential,
|
||||
hm.certVerifier(), func() (uint32, error) { return hm.allocateIndex(hh) },
|
||||
true, header.HandshakeIXPSK0,
|
||||
hm.laneAdvert(),
|
||||
)
|
||||
if err != nil {
|
||||
hm.f.l.Error("Failed to create handshake machine",
|
||||
@@ -695,6 +707,35 @@ func (hm *HandshakeManager) buildStage0Packet(hh *HandshakeHostInfo) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// laneAdvert returns our multiport advert for a handshake payload, or nil
|
||||
// when multiport is disabled (which keeps the payload byte-identical to
|
||||
// vanilla).
|
||||
func (hm *HandshakeManager) laneAdvert() *handshake.LaneDetails {
|
||||
if hm.config.laneCount == 0 {
|
||||
return nil
|
||||
}
|
||||
return &handshake.LaneDetails{
|
||||
PortCount: uint32(hm.config.lanePortCount),
|
||||
BasePort: uint32(hm.config.laneBasePort),
|
||||
TxLanes: uint32(hm.config.laneCount),
|
||||
}
|
||||
}
|
||||
|
||||
// maybeAllocLanes sets up the multiport lanes for a just-completed tunnel. Must
|
||||
// run before the hostinfo becomes visible in the hostmap: the data plane reads
|
||||
// hostinfo.lanes without synchronizing on it. The sessions themselves are derived
|
||||
// later, on the first packet that needs each one.
|
||||
func (hm *HandshakeManager) maybeAllocLanes(hostinfo *HostInfo, result *handshake.Result) {
|
||||
if hm.config.laneCount == 0 || result.PeerPortCount == 0 {
|
||||
return
|
||||
}
|
||||
if len(hm.f.myVpnAddrs) == 0 || len(hostinfo.vpnAddrs) == 0 {
|
||||
return
|
||||
}
|
||||
|
||||
hostinfo.lanes = newLaneSet(result, hm.config.laneCount, hm.f.myVpnAddrs[0], hostinfo.vpnAddrs[0])
|
||||
}
|
||||
|
||||
// beginHandshake handles an incoming handshake packet that doesn't match any
|
||||
// existing pending handshake. It creates a new responder Machine and processes
|
||||
// the first message.
|
||||
@@ -713,6 +754,7 @@ func (hm *HandshakeManager) beginHandshake(via ViaSender, packet []byte, h *head
|
||||
v, cs.GetCredential,
|
||||
hm.certVerifier(), func() (uint32, error) { return generateIndex(f.l) },
|
||||
false, header.HandshakeIXPSK0,
|
||||
hm.laneAdvert(),
|
||||
)
|
||||
if err != nil {
|
||||
f.l.Error("Failed to create handshake machine", "from", via, "error", err)
|
||||
@@ -769,6 +811,10 @@ func (hm *HandshakeManager) beginHandshake(via ViaSender, packet []byte, h *head
|
||||
},
|
||||
}
|
||||
|
||||
// Lanes are allocated before the log line so it can report what was actually
|
||||
// negotiated, and must in any case be in place before CheckAndComplete below.
|
||||
hm.maybeAllocLanes(hostinfo, result)
|
||||
|
||||
msg := "Handshake message received"
|
||||
if !anyVpnAddrsInCommon {
|
||||
msg = "Handshake message received, but no vpnNetworks in common."
|
||||
@@ -783,6 +829,7 @@ func (hm *HandshakeManager) beginHandshake(via ViaSender, packet []byte, h *head
|
||||
"initiatorIndex", result.RemoteIndex,
|
||||
"responderIndex", result.LocalIndex,
|
||||
"handshake", m{"stage": uint64(machine.MessageIndex()), "style": header.SubTypeName(header.Handshake, machine.Subtype())},
|
||||
laneLogAttr(hm.config.laneCount, hostinfo.lanes),
|
||||
)
|
||||
|
||||
// packet aliases the listener's incoming buffer, so this copy must stay.
|
||||
@@ -957,6 +1004,14 @@ func (hm *HandshakeManager) continueHandshake(via ViaSender, hh *HandshakeHostIn
|
||||
}
|
||||
|
||||
duration := time.Since(hh.startTime).Nanoseconds()
|
||||
|
||||
hostinfo.vpnAddrs = vpnAddrs
|
||||
hostinfo.buildNetworks(f.myVpnNetworksTable, remoteCert.Certificate)
|
||||
|
||||
// Lanes are allocated before the log line so it can report what was actually
|
||||
// negotiated, and must in any case be in place before Complete below.
|
||||
hm.maybeAllocLanes(hostinfo, result)
|
||||
|
||||
msg := "Handshake message received"
|
||||
if !anyVpnAddrsInCommon {
|
||||
msg = "Handshake message received, but no vpnNetworks in common."
|
||||
@@ -973,11 +1028,9 @@ func (hm *HandshakeManager) continueHandshake(via ViaSender, hh *HandshakeHostIn
|
||||
"handshake", m{"stage": uint64(machine.MessageIndex()), "style": header.SubTypeName(header.Handshake, machine.Subtype())},
|
||||
"durationNs", duration,
|
||||
"sentCachedPackets", len(hh.packetStore),
|
||||
laneLogAttr(hm.config.laneCount, hostinfo.lanes),
|
||||
)
|
||||
|
||||
hostinfo.vpnAddrs = vpnAddrs
|
||||
hostinfo.buildNetworks(f.myVpnNetworksTable, remoteCert.Certificate)
|
||||
|
||||
hm.Complete(hostinfo, f)
|
||||
|
||||
if len(hh.packetStore) > 0 {
|
||||
@@ -1086,7 +1139,10 @@ func (hm *HandshakeManager) sendHandshakeResponse(via ViaSender, msg []byte, hos
|
||||
|
||||
if !via.IsRelayed {
|
||||
fields := append(logFields, "from", via)
|
||||
err := f.outside.WriteTo(msg, via.UdpAddr)
|
||||
// Reply from the socket the handshake arrived on so the initiator sees
|
||||
// the source port it targeted. Identical to f.outside under vanilla
|
||||
// config (all writers share one port); required for multiport lanes.
|
||||
err := f.writers[via.SockIdx].WriteTo(msg, via.UdpAddr)
|
||||
if err != nil {
|
||||
f.l.Error("Failed to send handshake message", append(fields, "error", err)...)
|
||||
} else {
|
||||
|
||||
+51
-15
@@ -8,16 +8,28 @@ import (
|
||||
)
|
||||
|
||||
//Version 1 header:
|
||||
// 0 31
|
||||
// |-----------------------------------------------------------------------|
|
||||
// | Version (uint4) | Type (uint4) | Subtype (uint8) | Reserved (uint16) | 32
|
||||
// |-----------------------------------------------------------------------|
|
||||
// | Remote index (uint32) | 64
|
||||
// |-----------------------------------------------------------------------|
|
||||
// | Message counter | 96
|
||||
// | (uint64) | 128
|
||||
// |-----------------------------------------------------------------------|
|
||||
// | payload... |
|
||||
// 0 31
|
||||
// |------------------------------------------------------------------------------------|
|
||||
// | Version (uint4) | Type (uint4) | Subtype (uint8) | Reserved (uint8) | Lane (uint8) | 32
|
||||
// |------------------------------------------------------------------------------------|
|
||||
// | Remote index (uint32) | 64
|
||||
// |------------------------------------------------------------------------------------|
|
||||
// | Message counter | 96
|
||||
// | (uint64) | 128
|
||||
// |------------------------------------------------------------------------------------|
|
||||
// | payload... |
|
||||
//
|
||||
// Lane is the multiport lane index, carved out of the low 8 bits of what was a
|
||||
// single Reserved (uint16) before multiport. Lane 0 is the base tunnel, which is
|
||||
// what every sender that does not know about lanes emits and what every non-lane
|
||||
// packet — handshakes, lighthouse, relays, close — carries, so the field is
|
||||
// compatible in both directions with a peer that has never heard of it. The
|
||||
// remaining 8 bits stay reserved and are always sent as zero. The H struct still
|
||||
// holds the pair as one Reserved field; use H.Lane and EncodeLane to reach the
|
||||
// low byte.
|
||||
//
|
||||
// Lane is part of the AEAD's associated data, so a lane index cannot be altered
|
||||
// in flight: a packet decrypts on the lane it claims or not at all.
|
||||
|
||||
type m = map[string]any
|
||||
|
||||
@@ -57,8 +69,18 @@ const (
|
||||
const (
|
||||
TestRequest MessageSubType = 0
|
||||
TestReply MessageSubType = 1
|
||||
// LaneProbe is sent on a multiport lane to prove the lane's 5-tuple is
|
||||
// usable; LaneProbeAck answers it on the base tunnel.
|
||||
LaneProbe MessageSubType = 2
|
||||
LaneProbeAck MessageSubType = 3
|
||||
)
|
||||
|
||||
// MaxLane is the largest lane index the header can carry.
|
||||
const MaxLane = 0xff
|
||||
|
||||
// laneMask covers the bits of Reserved that hold the lane index.
|
||||
const laneMask uint16 = 0x00ff
|
||||
|
||||
const (
|
||||
HandshakeIXPSK0 MessageSubType = 0
|
||||
HandshakeXXPSK0 MessageSubType = 1
|
||||
@@ -67,8 +89,10 @@ const (
|
||||
var ErrHeaderTooShort = errors.New("header is too short")
|
||||
|
||||
var subTypeTestMap = map[MessageSubType]string{
|
||||
TestRequest: "testRequest",
|
||||
TestReply: "testReply",
|
||||
TestRequest: "testRequest",
|
||||
TestReply: "testReply",
|
||||
LaneProbe: "laneProbe",
|
||||
LaneProbeAck: "laneProbeAck",
|
||||
}
|
||||
|
||||
var subTypeNoneMap = map[MessageSubType]string{0: "none"}
|
||||
@@ -100,10 +124,16 @@ type H struct {
|
||||
// Encode uses the provided byte array to encode the provided header values into.
|
||||
// Byte array must be capped higher than HeaderLen or this will panic
|
||||
func Encode(b []byte, v uint8, t MessageType, st MessageSubType, ri uint32, c uint64) []byte {
|
||||
return EncodeLane(b, v, t, st, ri, c, 0)
|
||||
}
|
||||
|
||||
// EncodeLane is Encode with an explicit multiport lane index, which is carried
|
||||
// in the low 8 bits of Reserved.
|
||||
func EncodeLane(b []byte, v uint8, t MessageType, st MessageSubType, ri uint32, c uint64, lane uint8) []byte {
|
||||
b = b[:Len]
|
||||
b[0] = v<<4 | byte(t&0x0f)
|
||||
b[1] = byte(st)
|
||||
binary.BigEndian.PutUint16(b[2:4], 0)
|
||||
binary.BigEndian.PutUint16(b[2:4], uint16(lane))
|
||||
binary.BigEndian.PutUint32(b[4:8], ri)
|
||||
binary.BigEndian.PutUint64(b[8:16], c)
|
||||
return b
|
||||
@@ -136,7 +166,13 @@ func (h *H) Encode(b []byte) ([]byte, error) {
|
||||
return nil, errors.New("nil header")
|
||||
}
|
||||
|
||||
return Encode(b, h.Version, h.Type, h.Subtype, h.RemoteIndex, h.MessageCounter), nil
|
||||
return EncodeLane(b, h.Version, h.Type, h.Subtype, h.RemoteIndex, h.MessageCounter, h.Lane()), nil
|
||||
}
|
||||
|
||||
// Lane returns the multiport lane index carried in Reserved. Lane 0 is the base
|
||||
// tunnel, which is what any sender that does not know about lanes will report.
|
||||
func (h *H) Lane() uint8 {
|
||||
return uint8(h.Reserved & laneMask)
|
||||
}
|
||||
|
||||
// Parse is a helper function to parses given bytes into new Header struct
|
||||
@@ -196,7 +232,7 @@ func IsValidSubType(t MessageType, s MessageSubType) bool {
|
||||
case Handshake:
|
||||
return s == HandshakeIXPSK0
|
||||
case Test:
|
||||
return s == TestReply || s == TestRequest
|
||||
return s == TestReply || s == TestRequest || s == LaneProbe || s == LaneProbeAck
|
||||
case Control, CloseTunnel, RecvError, LightHouse:
|
||||
return s == 0
|
||||
default:
|
||||
|
||||
+25
-1
@@ -52,6 +52,28 @@ func TestParse(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestEncodeLane(t *testing.T) {
|
||||
b := EncodeLane(make([]byte, Len), Version, Message, MessageNone, 10, 9, 3)
|
||||
assert.Equal(t, []byte{0x11, 0x0, 0x0, 0x3}, b[:4])
|
||||
|
||||
h := &H{}
|
||||
require.NoError(t, h.Parse(b))
|
||||
assert.Equal(t, uint16(3), h.Reserved)
|
||||
assert.Equal(t, uint8(3), h.Lane())
|
||||
|
||||
// Encode is EncodeLane on the base tunnel, and the H method round trips the lane.
|
||||
assert.Equal(t,
|
||||
EncodeLane(make([]byte, Len), Version, Message, MessageNone, 10, 9, 0),
|
||||
Encode(make([]byte, Len), Version, Message, MessageNone, 10, 9))
|
||||
|
||||
rt, err := h.Encode(make([]byte, Len))
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, b, rt)
|
||||
|
||||
// Only the low 8 bits of Reserved are the lane.
|
||||
assert.Equal(t, uint8(0x2a), (&H{Reserved: 0xff2a}).Lane())
|
||||
}
|
||||
|
||||
func TestTypeName(t *testing.T) {
|
||||
assert.Equal(t, "test", TypeName(Test))
|
||||
assert.Equal(t, "test", (&H{Type: Test}).TypeName())
|
||||
@@ -127,7 +149,9 @@ func TestIsValidSubType(t *testing.T) {
|
||||
|
||||
assert.True(t, IsValidSubType(Test, TestRequest))
|
||||
assert.True(t, IsValidSubType(Test, TestReply))
|
||||
assert.False(t, IsValidSubType(Test, 2))
|
||||
assert.True(t, IsValidSubType(Test, LaneProbe))
|
||||
assert.True(t, IsValidSubType(Test, LaneProbeAck))
|
||||
assert.False(t, IsValidSubType(Test, 4))
|
||||
|
||||
// These types only ever carry subtype 0.
|
||||
for _, mt := range []MessageType{Control, CloseTunnel, RecvError, LightHouse} {
|
||||
|
||||
+13
@@ -278,6 +278,11 @@ type HostInfo struct {
|
||||
// This value will be behind against actual tunnel utilization in the hot path.
|
||||
// This should only be used by the ConnectionManagers ticker routine.
|
||||
lastUsed time.Time
|
||||
|
||||
// lanes holds this tunnel's multiport lane sessions. Allocated when the
|
||||
// handshake completes if both sides advertised multiport, nil otherwise.
|
||||
// Immutable once the hostinfo is published to the data plane.
|
||||
lanes *laneSet
|
||||
}
|
||||
|
||||
type ViaSender struct {
|
||||
@@ -285,6 +290,11 @@ type ViaSender struct {
|
||||
relayHI *HostInfo // relayHI is the host info object of the relay
|
||||
relay *Relay // relay contains the rest of the relay information, including the PeerIP of the host trying to communicate with us.
|
||||
IsRelayed bool // IsRelayed is true if the packet was sent through a relay
|
||||
|
||||
// SockIdx is the local socket (Interface.writers index) the packet
|
||||
// arrived on. Replies that must originate from the same 4-tuple egress
|
||||
// f.writers[SockIdx].
|
||||
SockIdx int
|
||||
}
|
||||
|
||||
func (v ViaSender) String() string {
|
||||
@@ -474,6 +484,9 @@ func (hm *HostMap) unlockedMakePrimary(hostinfo *HostInfo) bool {
|
||||
// any tunnel to the peer), which the caller uses to decide whether to clear learned lighthouse
|
||||
// state and disestablish relays.
|
||||
func (hm *HostMap) unlockedDeleteHostInfo(hostinfo *HostInfo) bool {
|
||||
// Lane sessions hang off this hostinfo, so deleting it takes them with it
|
||||
// and there is nothing extra to unwind here.
|
||||
|
||||
// Remove this hostinfo from each of its address lists. The lists are independent, so a
|
||||
// sibling is never promoted to an address it does not own and no other list is touched.
|
||||
final := true
|
||||
|
||||
@@ -11,12 +11,11 @@ import (
|
||||
"github.com/slackhq/nebula/header"
|
||||
"github.com/slackhq/nebula/iputil"
|
||||
"github.com/slackhq/nebula/noiseutil"
|
||||
"github.com/slackhq/nebula/overlay/batch"
|
||||
"github.com/slackhq/nebula/overlay/tio"
|
||||
"github.com/slackhq/nebula/routing"
|
||||
)
|
||||
|
||||
func (f *Interface) consumeInsidePacket(pkt tio.Packet, fwPacket *firewall.ParsedPacket, nb []byte, sendBatch *batch.SendBatch, rejectBuf []byte, q int, localCache firewall.ConntrackCache) {
|
||||
func (f *Interface) consumeInsidePacket(pkt tio.Packet, fwPacket *firewall.ParsedPacket, nb []byte, tx *txQueue, rejectBuf []byte, q int, localCache firewall.ConntrackCache) {
|
||||
// borrowed: pkt.Bytes is owned by the originating tio.Queue and is
|
||||
// only valid until the next Read on that queue. Every consumer below
|
||||
// (parse, self-forward, handshake cache, sendInsideMessage) reads it
|
||||
@@ -111,7 +110,7 @@ func (f *Interface) consumeInsidePacket(pkt tio.Packet, fwPacket *firewall.Parse
|
||||
|
||||
dropReason := f.firewall.Drop(fwPacket.Packet, false, hostinfo, f.pki.GetCAPool(), localCache)
|
||||
if dropReason == nil {
|
||||
f.sendInsideMessage(hostinfo, pkt, nb, sendBatch)
|
||||
f.sendInsideMessage(hostinfo, pkt, &fwPacket.Packet, nb, tx)
|
||||
} else {
|
||||
f.rejectInside(packet, rejectBuf, q)
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
@@ -123,13 +122,13 @@ func (f *Interface) consumeInsidePacket(pkt tio.Packet, fwPacket *firewall.Parse
|
||||
}
|
||||
}
|
||||
|
||||
func (f *Interface) sendInsideEncrypt(hostinfo *HostInfo, ci *ConnectionState, seg, scratch, nb []byte) []byte {
|
||||
func (f *Interface) sendInsideEncrypt(hostinfo *HostInfo, ci *ConnectionState, lane uint8, seg, scratch, nb []byte) []byte {
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
ci.writeLock.Lock()
|
||||
}
|
||||
c := ci.messageCounter.Add(1)
|
||||
|
||||
out := header.Encode(scratch, header.Version, header.Message, 0, hostinfo.remoteIndexId, c)
|
||||
out := header.EncodeLane(scratch, header.Version, header.Message, 0, hostinfo.remoteIndexId, c, lane)
|
||||
|
||||
out, encErr := ci.eKey.EncryptDanger(out, out, seg, c, nb)
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
@@ -154,12 +153,19 @@ func (f *Interface) sendInsideEncrypt(hostinfo *HostInfo, ci *ConnectionState, s
|
||||
// kernel-supplied superpacket bytes never get written into a separate
|
||||
// scratch arena: SegmentSuperpacket builds each segment's plaintext in
|
||||
// segScratch[:segLen] in turn, and we encrypt directly into a fresh SendBatch slot.
|
||||
func (f *Interface) sendInsideMessage(hostinfo *HostInfo, pkt tio.Packet, nb []byte, sendBatch *batch.SendBatch) {
|
||||
//
|
||||
// When this flow has a usable multiport lane to this peer, the direct path swaps
|
||||
// to that lane's session and socket below. Relay and base traffic stays on
|
||||
// tx.base (socket 0).
|
||||
func (f *Interface) sendInsideMessage(hostinfo *HostInfo, pkt tio.Packet, fwPacket *firewall.Packet, nb []byte, tx *txQueue) {
|
||||
ci := hostinfo.ConnectionState
|
||||
if ci.eKey == nil {
|
||||
return
|
||||
}
|
||||
|
||||
// Base and relay traffic stays on socket 0; the direct path may swap to tx.lane below.
|
||||
sendBatch := tx.base
|
||||
|
||||
// One traffic-out mark covers every segment of the superpacket; doing it
|
||||
// per segment in sendInsideEncrypt paid an atomic store up to ~45 extra
|
||||
// times per TSO packet, inside writeLock under boring crypto.
|
||||
@@ -201,7 +207,7 @@ func (f *Interface) sendInsideMessage(hostinfo *HostInfo, pkt tio.Packet, nb []b
|
||||
//relay header + header + plaintext + AEAD tag (16 bytes for both AES-GCM and ChaCha20-Poly1305) + relay tag
|
||||
scratch := sendBatch.Reserve(header.Len + header.Len + len(seg) + 16 + 16)
|
||||
|
||||
innerPacket := f.sendInsideEncrypt(hostinfo, ci, seg, scratch[header.Len:], nb)
|
||||
innerPacket := f.sendInsideEncrypt(hostinfo, ci, 0, seg, scratch[header.Len:], nb)
|
||||
if innerPacket == nil {
|
||||
return nil
|
||||
}
|
||||
@@ -222,11 +228,28 @@ func (f *Interface) sendInsideMessage(hostinfo *HostInfo, pkt tio.Packet, nb []b
|
||||
return
|
||||
}
|
||||
|
||||
// Direct path: prefer this flow's multiport lane once it is proven usable.
|
||||
// txLaneForFlow hands back the lane's session and destination together, so
|
||||
// there is no window where one is set and the other is not, and a demotion
|
||||
// drops us back onto the base tunnel on the very next packet.
|
||||
//
|
||||
// A miss is also how a lane gets re-probed after a demotion: txLane raises
|
||||
// demand, which the connection manager's next tick on this tunnel picks up.
|
||||
// Until the lane is up the traffic rides the base tunnel, the same fallback
|
||||
// a demoted lane uses.
|
||||
lane := uint8(0)
|
||||
if s, lci, laneRemote := hostinfo.lanes.txLaneForFlow(fwPacket); lci != nil {
|
||||
lane = uint8(s)
|
||||
ci = lci
|
||||
remote = laneRemote
|
||||
sendBatch = tx.laneBatch(f, s)
|
||||
}
|
||||
|
||||
err := tio.SegmentSuperpacket(pkt, func(seg []byte) error {
|
||||
// header + plaintext + AEAD tag (16 bytes for both AES-GCM and ChaCha20-Poly1305)
|
||||
scratch := sendBatch.Reserve(header.Len + len(seg) + 16)
|
||||
|
||||
out := f.sendInsideEncrypt(hostinfo, ci, seg, scratch, nb)
|
||||
out := f.sendInsideEncrypt(hostinfo, ci, lane, seg, scratch, nb)
|
||||
if out == nil {
|
||||
return nil
|
||||
}
|
||||
@@ -515,16 +538,50 @@ func (f *Interface) SendVia(via *HostInfo, relay *Relay, ad, nb, out []byte, noc
|
||||
return
|
||||
}
|
||||
|
||||
err = f.writers[q].WriteTo(toSend, via.GetRemote())
|
||||
err = f.writers[f.egressSock(q)].WriteTo(toSend, via.GetRemote())
|
||||
if err != nil {
|
||||
via.logger(f.l).Info("Failed to WriteTo in sendVia", "error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// egressSock picks the socket a tunnel packet leaves from.
|
||||
//
|
||||
// Everything that is not lane data plane leaves from the base port: handshakes, keepalives, close packets, rejects and
|
||||
// relay carriers all belong to the base tunnel's 4-tuple, which is the only one a peer's spoof/roam checks and a
|
||||
// vanilla peer's expectations know about. Lane data goes through laneSock instead and never comes here.
|
||||
//
|
||||
// Which socket on the base port doesn't matter — they share an address, so they produce identical packets — so keep to
|
||||
// this routine's own share of the group and leave the rest of it uncontended. Without multiport that is q itself, since
|
||||
// every socket is on the base port.
|
||||
func (f *Interface) egressSock(q int) int {
|
||||
return f.laneSock(q, 0)
|
||||
}
|
||||
|
||||
// laneSock returns the index in writers of a socket bound to lane s's port, for a
|
||||
// routine that reads queue q.
|
||||
//
|
||||
// Under multiport the sockets are laid out port-major — writers[s*routinesPerPort
|
||||
// + r] is the r'th socket on port listen.port+s — so every routine has a sibling
|
||||
// socket on every port and the arithmetic is a lane index away. Routines pick the
|
||||
// sibling matching their own position in their group, which spreads the writers
|
||||
// for one port over that port's whole group rather than funnelling them onto its
|
||||
// first socket. It is a pure function of (q, s), so a flow always leaves from the
|
||||
// same socket and cannot reorder itself across two of them.
|
||||
//
|
||||
// Without multiport there is one port and every socket is on it, so any lane
|
||||
// resolves to q's own socket.
|
||||
func (f *Interface) laneSock(q, s int) int {
|
||||
if !f.multiport {
|
||||
return q
|
||||
}
|
||||
return s*f.routinesPerPort + q%f.routinesPerPort
|
||||
}
|
||||
|
||||
func (f *Interface) sendNoMetrics(t header.MessageType, st header.MessageSubType, ci *ConnectionState, hostinfo *HostInfo, remote netip.AddrPort, p, nb, out []byte, q int) {
|
||||
if ci.eKey == nil {
|
||||
return
|
||||
}
|
||||
q = f.egressSock(q)
|
||||
useRelay := !remote.IsValid() && !hostinfo.GetRemote().IsValid()
|
||||
fullOut := out
|
||||
|
||||
|
||||
+157
-12
@@ -43,10 +43,22 @@ type InterfaceConfig struct {
|
||||
DropLocalBroadcast bool
|
||||
DropMulticast bool
|
||||
routines int
|
||||
MessageMetrics *MessageMetrics
|
||||
version string
|
||||
relayManager *relayManager
|
||||
punchy *Punchy
|
||||
// Multiport means the sockets are spread over a range of ports
|
||||
// (listen.port+slot) rather than all sharing listen.port, and that lane
|
||||
// tunnels are negotiated with capable peers.
|
||||
Multiport bool
|
||||
// RoutinesPerPort is how many sockets share each port under multiport, and so
|
||||
// the stride between port slots in writers: writers[s*RoutinesPerPort+r] is
|
||||
// the r'th socket bound to listen.port+s. It is `routines` as configured,
|
||||
// while routines above is that times the number of ports.
|
||||
RoutinesPerPort int
|
||||
// LaneCount is the number of lanes counting the base tunnel as lane 0
|
||||
// (multiport.lanes, clamped to the number of ports bound).
|
||||
LaneCount int
|
||||
MessageMetrics *MessageMetrics
|
||||
version string
|
||||
relayManager *relayManager
|
||||
punchy *Punchy
|
||||
|
||||
tryPromoteEvery uint32
|
||||
reQueryEvery uint32
|
||||
@@ -87,6 +99,9 @@ type Interface struct {
|
||||
dropLocalBroadcast bool
|
||||
dropMulticast bool
|
||||
routines int
|
||||
multiport bool
|
||||
routinesPerPort int
|
||||
laneCount int
|
||||
disconnectInvalid atomic.Bool
|
||||
closed atomic.Bool
|
||||
// cpuAffinity, when non-empty, names the CPUs each TUN reader goroutine
|
||||
@@ -217,6 +232,9 @@ func NewInterface(ctx context.Context, c *InterfaceConfig) (*Interface, error) {
|
||||
dropLocalBroadcast: c.DropLocalBroadcast,
|
||||
dropMulticast: c.DropMulticast,
|
||||
routines: c.routines,
|
||||
multiport: c.Multiport,
|
||||
routinesPerPort: max(c.RoutinesPerPort, 1),
|
||||
laneCount: c.LaneCount,
|
||||
version: c.version,
|
||||
writers: make([]udp.Conn, c.routines),
|
||||
batchers: make([]*batch.MultiCoalescer, c.routines),
|
||||
@@ -276,7 +294,10 @@ func (f *Interface) activate() error {
|
||||
"fips140Enforced", fips140.Enforced(),
|
||||
)
|
||||
|
||||
if f.routines > 1 && !f.outside.SupportsMultipleReaders() {
|
||||
// Under multiport each socket has exactly one reader on its own port, so
|
||||
// the shared-port multi-reader capability is irrelevant (and main.go
|
||||
// already hard-errored on unsupported platforms).
|
||||
if f.routines > 1 && !f.multiport && !f.outside.SupportsMultipleReaders() {
|
||||
f.routines = 1
|
||||
f.l.Warn("multiple udp readers are not supported on this platform, falling back to a single routine")
|
||||
}
|
||||
@@ -289,6 +310,11 @@ func (f *Interface) activate() error {
|
||||
return err
|
||||
}
|
||||
if len(queues) < f.routines {
|
||||
if f.multiport {
|
||||
// The lane sockets are already bound one-per-routine; shrinking
|
||||
// the routine count would leave bound ports with no reader.
|
||||
return fmt.Errorf("multiport requires %d tun queues, device provided %d", f.routines, len(queues))
|
||||
}
|
||||
// TODO: this clamp is only safe because it is unreachable when the
|
||||
// udp side has multiple readers (linux Queues opens exactly n or
|
||||
// errors; every other platform already clamped routines to 1 above).
|
||||
@@ -389,7 +415,7 @@ func (f *Interface) listenOut(i int) {
|
||||
rxc := newRxContext(f, i)
|
||||
|
||||
listener := func(fromUdpAddr netip.AddrPort, payload []byte) {
|
||||
f.readOutsidePackets(ViaSender{UdpAddr: fromUdpAddr}, payload, rxc)
|
||||
f.readOutsidePackets(ViaSender{UdpAddr: fromUdpAddr, SockIdx: i}, payload, rxc)
|
||||
}
|
||||
|
||||
flusher := func() {
|
||||
@@ -431,6 +457,114 @@ func (f *Interface) pinThisThread(i int) {
|
||||
}
|
||||
}
|
||||
|
||||
// txQueue is the per-routine TX state owned by one listenIn goroutine.
|
||||
//
|
||||
// base carries base-session data, relay carriers, and everything on a tunnel
|
||||
// without lanes. It goes out a socket on the base port — egressSock's pick — since
|
||||
// base traffic must keep the base source port or a vanilla peer would see it move
|
||||
// and roam-thrash. lane[s] goes out a socket on listen.port+s and carries traffic
|
||||
// encrypted with lane s's session; lane[0] is base, and the rest are built on the
|
||||
// first packet that picks them, since a routine that never sends on a lane should
|
||||
// not hold a batch for it. Both come from laneSock, so a routine writes to its own
|
||||
// share of each port's socket group.
|
||||
//
|
||||
// Which lane a packet rides comes from its own flow hash, not from this
|
||||
// routine's index. That is deliberate. Which routine reads a flow is the
|
||||
// kernel's decision: it hashes the flow to a tun queue, but it also *learns*
|
||||
// the queue we write that flow's inbound packets to, and prefers what it
|
||||
// learned. So if the lane followed the routine, a peer whose lanes were still
|
||||
// down — every peer, for the first moments of a tunnel — would write all of its
|
||||
// inbound traffic to queue 0, teaching both kernels to steer every flow to
|
||||
// queue 0, and every tunnel would collapse onto lane 0 and stay there for as
|
||||
// long as its flows kept busy. Hashing here makes lane spread independent of
|
||||
// tun steering entirely.
|
||||
//
|
||||
// Every batch borrows arena, so a routine holding a batch per lane still costs
|
||||
// one slab. The arena is reset by flush once every batch over it is drained.
|
||||
//
|
||||
// Several routines can still write to one socket — the sockets on a port are
|
||||
// shared by the routines whose lane arithmetic lands on them — which the underlay
|
||||
// serializes (see batchWriter). Per-flow wire order still holds: a flow is hashed
|
||||
// onto one lane and read by one routine, so nothing else is writing it.
|
||||
type txQueue struct {
|
||||
// q is the queue this state belongs to, which laneSock needs to resolve a lane
|
||||
// to one of its port's sockets.
|
||||
q int
|
||||
base *batch.SendBatch
|
||||
lane []*batch.SendBatch
|
||||
arena *batch.Arena
|
||||
|
||||
// live is every batch built so far, in build order, so base is first: see
|
||||
// flush. Kept as its own slice because lane is mostly nil holes and both
|
||||
// full and flush walk this per read batch.
|
||||
live []txBatch
|
||||
}
|
||||
|
||||
// txBatch is a live batch and the index in writers of the socket it flushes to.
|
||||
type txBatch struct {
|
||||
sb *batch.SendBatch
|
||||
sock int
|
||||
}
|
||||
|
||||
func (f *Interface) newTxQueue(q int) *txQueue {
|
||||
baseSock := f.egressSock(q)
|
||||
arena := batch.NewArena(batch.SendBatchCap * (udp.MTU + 32))
|
||||
base := batch.NewSendBatchSharedArena(f.writers[baseSock], batch.SendBatchCap, arena)
|
||||
|
||||
tx := &txQueue{
|
||||
q: q,
|
||||
base: base,
|
||||
arena: arena,
|
||||
live: []txBatch{{sb: base, sock: baseSock}},
|
||||
}
|
||||
if f.multiport && f.laneCount > 1 {
|
||||
tx.lane = make([]*batch.SendBatch, f.laneCount)
|
||||
tx.lane[0] = base
|
||||
}
|
||||
return tx
|
||||
}
|
||||
|
||||
// laneBatch returns the batch for lane s, building it the first time this
|
||||
// routine sends on that lane. Lanes this queue doesn't cover fall back to base,
|
||||
// which is also lane 0's batch.
|
||||
func (tx *txQueue) laneBatch(f *Interface, s int) *batch.SendBatch {
|
||||
if s <= 0 || s >= len(tx.lane) {
|
||||
return tx.base
|
||||
}
|
||||
sb := tx.lane[s]
|
||||
if sb == nil {
|
||||
sock := f.laneSock(tx.q, s)
|
||||
sb = batch.NewSendBatchSharedArena(f.writers[sock], batch.SendBatchCap, tx.arena)
|
||||
tx.lane[s] = sb
|
||||
tx.live = append(tx.live, txBatch{sb: sb, sock: sock})
|
||||
}
|
||||
return sb
|
||||
}
|
||||
|
||||
// full reports a full sendmmsg worth of work queued across every lane, rather
|
||||
// than on any one of them: the arena is shared, so it is the total that bounds
|
||||
// how much is outstanding.
|
||||
func (tx *txQueue) full() bool {
|
||||
n := 0
|
||||
for _, b := range tx.live {
|
||||
n += b.sb.Len()
|
||||
}
|
||||
return n >= batch.SendBatchCap
|
||||
}
|
||||
|
||||
// flush drains base before the lanes so that when a flow moves from the base
|
||||
// session onto a freshly promoted lane mid-window, its packets still leave this
|
||||
// host in encryption order. Resetting the shared arena is this queue's job,
|
||||
// since no single batch's Flush can know the others are done with it.
|
||||
func (tx *txQueue) flush(f *Interface) {
|
||||
for _, b := range tx.live {
|
||||
if b.sb.Len() > 0 {
|
||||
f.flushSendBatch(b.sb, b.sock)
|
||||
}
|
||||
}
|
||||
tx.arena.Reset()
|
||||
}
|
||||
|
||||
func (f *Interface) listenIn(queue tio.Queue, i int) {
|
||||
// Pinning this thread (and goroutine) to a single CPU keeps every sendmmsg from this goroutine going through the
|
||||
// same TX ring on the nic, so the wire sees per-flow order. Skip entirely when tun.pin_threads is false.
|
||||
@@ -439,8 +573,7 @@ func (f *Interface) listenIn(queue tio.Queue, i int) {
|
||||
}
|
||||
|
||||
rejectBuf := make([]byte, mtu)
|
||||
arenaSize := batch.SendBatchCap * (udp.MTU + 32)
|
||||
sb := batch.NewSendBatch(f.writers[i], batch.SendBatchCap, arenaSize)
|
||||
tx := f.newTxQueue(i)
|
||||
fwPacket := &firewall.ParsedPacket{}
|
||||
nb := make([]byte, 12, 12)
|
||||
|
||||
@@ -458,15 +591,15 @@ func (f *Interface) listenIn(queue tio.Queue, i int) {
|
||||
}
|
||||
|
||||
for _, pkt := range pkts {
|
||||
f.consumeInsidePacket(pkt, fwPacket, nb, sb, rejectBuf, i, conntrackCache.Get())
|
||||
f.consumeInsidePacket(pkt, fwPacket, nb, tx, rejectBuf, i, conntrackCache.Get())
|
||||
// Flush incrementally once a full sendmmsg batch has
|
||||
// accumulated so the first packets of a deep read drain
|
||||
// hit the wire while the rest are still being encrypted.
|
||||
if sb.Len() >= batch.SendBatchCap {
|
||||
f.flushSendBatch(sb, i)
|
||||
if tx.full() {
|
||||
tx.flush(f)
|
||||
}
|
||||
}
|
||||
f.flushSendBatch(sb, i)
|
||||
tx.flush(f)
|
||||
}
|
||||
|
||||
f.l.Debug("overlay reader is done", "reader", i)
|
||||
@@ -634,11 +767,23 @@ func (f *Interface) emitStats(ctx context.Context, i time.Duration) {
|
||||
certInitiatingVersion := metrics.GetOrRegisterGauge("certificate.initiating_version", nil)
|
||||
certMaxVersion := metrics.GetOrRegisterGauge("certificate.max_version", nil)
|
||||
|
||||
// Registered only when we run multiport, so these don't sit at zero on a node
|
||||
// that was never going to have a lane and read as a broken feature.
|
||||
var lanesUpGauge, laneTunnelsGauge metrics.Gauge
|
||||
if f.multiport && f.laneCount > 1 {
|
||||
lanesUpGauge = metrics.GetOrRegisterGauge("multiport.lanes.up", nil)
|
||||
laneTunnelsGauge = metrics.GetOrRegisterGauge("multiport.lanes.tunnels", nil)
|
||||
}
|
||||
|
||||
emit := func() {
|
||||
f.firewall.EmitStats()
|
||||
f.handshakeManager.EmitStats()
|
||||
udpStats()
|
||||
|
||||
if lanesUpGauge != nil {
|
||||
f.emitLaneStats(lanesUpGauge, laneTunnelsGauge)
|
||||
}
|
||||
|
||||
certState := f.pki.getCertState()
|
||||
defaultCrt := certState.GetDefaultCertificate()
|
||||
certExpirationGauge.Update(int64(defaultCrt.NotAfter().Sub(time.Now()) / time.Second))
|
||||
|
||||
@@ -0,0 +1,713 @@
|
||||
package nebula
|
||||
|
||||
import (
|
||||
"context"
|
||||
"hash/fnv"
|
||||
"log/slog"
|
||||
"net/netip"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/flynn/noise"
|
||||
"github.com/rcrowley/go-metrics"
|
||||
"github.com/slackhq/nebula/cert"
|
||||
"github.com/slackhq/nebula/firewall"
|
||||
"github.com/slackhq/nebula/handshake"
|
||||
"github.com/slackhq/nebula/header"
|
||||
"github.com/slackhq/nebula/noiseutil"
|
||||
)
|
||||
|
||||
// Multiport lanes give one tunnel several underlay 5-tuples, so its traffic
|
||||
// spreads over ECMP paths, NIC receive queues and per-flow policers instead of
|
||||
// funnelling through a single flow. Each inside flow picks a lane by hashing its
|
||||
// own 5-tuple, so the spread doesn't depend on how a kernel steers tun queues
|
||||
// (see txQueue).
|
||||
//
|
||||
// A lane is not a second tunnel: it is an extra session on the same HostInfo.
|
||||
// Noise leaves us with A.eKey == B.dKey, so both sides expand the same two keys
|
||||
// with the same per-lane label and land on a matched pair without exchanging
|
||||
// anything. A lane therefore costs no handshake, has no half-established state,
|
||||
// and dies exactly when its base tunnel does. Which lane a packet belongs to
|
||||
// travels in the nebula header, inside the AEAD's associated data.
|
||||
//
|
||||
// Lane 0 is the base tunnel itself: HostInfo.ConnectionState, the base port, and
|
||||
// the peer's real remote address, and it carries its share of flows like any
|
||||
// other. Lane s > 0 egresses a socket on listen.port+s (see laneSock) toward the
|
||||
// peer's advertised port range. Receiving on a lane
|
||||
// needs no permission — the keys are derivable the moment the base handshake
|
||||
// completes — but sending on one needs proof the new 5-tuple actually works,
|
||||
// since nothing else would notice a middlebox quietly dropping it. So a lane
|
||||
// stays down until a probe on it is acked, and falls back to the base tunnel the
|
||||
// moment it stops being acked.
|
||||
//
|
||||
// Both directions are pay-per-use. A lane session is derived on the first packet
|
||||
// that needs it, because how many lanes exist is partly the peer's call: it
|
||||
// advertises how many it sends on, and we have to be able to receive all of
|
||||
// them. Deriving them all up front would let a peer advertising the maximum cost
|
||||
// us a replay window and two cipher states per lane, per tunnel, for lanes it
|
||||
// may never send on.
|
||||
|
||||
const (
|
||||
// laneKeyInfo is the HKDF label prefix for lane key expansion. Changing it
|
||||
// means older builds derive different keys and drop our lane traffic; the
|
||||
// base tunnel would keep working, so the failure would be a silent loss of
|
||||
// lanes rather than of connectivity.
|
||||
laneKeyInfo = "nebula multiport lane v1"
|
||||
|
||||
// laneRetryBase and laneRetryMax bound the backoff between probes of a lane
|
||||
// that will not come up, so a peer whose lane ports are firewalled costs one
|
||||
// packet a minute rather than one per traffic tick.
|
||||
laneRetryBase = 5 * time.Second
|
||||
laneRetryMax = 60 * time.Second
|
||||
|
||||
// laneMaxFails caps the failure counter; the backoff saturates well before.
|
||||
laneMaxFails = 8
|
||||
|
||||
// laneProbeTimeout is how long a probe may go unacked before it counts as a
|
||||
// failure. It is shorter than the connection manager's check interval on
|
||||
// purpose: an outstanding probe is judged on the next tick either way, and a
|
||||
// longer timeout would only delay that by a whole tick.
|
||||
laneProbeTimeout = 2 * time.Second
|
||||
|
||||
// laneKeepalive is how often a lane that is up re-proves its path. Traffic
|
||||
// on a lane is not evidence the lane works — that is the whole reason lanes
|
||||
// need probing — so a lane that silently breaks is only caught here.
|
||||
laneKeepalive = 30 * time.Second
|
||||
)
|
||||
|
||||
// laneSet holds a peer's lane sessions and the state deciding which lanes may
|
||||
// carry traffic. It is built when the base handshake completes and never
|
||||
// resized, so the slices and their lengths are immutable; mu guards the fields
|
||||
// under it, and sessions/txAddr/demand are atomics read by the data plane
|
||||
// without it.
|
||||
type laneSet struct {
|
||||
// sessions[s] holds lane s's session once something has needed it, and nil
|
||||
// until then. sessions[0] is never populated: lane 0 is the base tunnel's own
|
||||
// ConnectionState. The length is immutable, so the data plane bounds-checks
|
||||
// and loads with no locking.
|
||||
sessions []atomic.Pointer[ConnectionState]
|
||||
|
||||
// material is what a lane session is derived from, kept because sessions are
|
||||
// derived lazily and the handshake result is long gone by then.
|
||||
material laneMaterial
|
||||
|
||||
// txAddr[s] holds lane s's remote address while the lane is proven usable
|
||||
// and nil otherwise. This single atomic is both the TX gate and the
|
||||
// destination, so a routine that loads non-nil has everything it needs.
|
||||
// Sized txLanes: lanes above that never send.
|
||||
txAddr []atomic.Pointer[netip.AddrPort]
|
||||
|
||||
// demand[s] is raised at creation, and again by the TX path whenever a flow
|
||||
// hashes onto lane s while it is down. Probing is demand-driven, so a peer we
|
||||
// never send to costs nothing beyond its base tunnel no matter how many lanes
|
||||
// are configured, and a lane that keeps failing is only retried while
|
||||
// something still wants it. Sized txLanes.
|
||||
demand []atomic.Bool
|
||||
|
||||
// txLanes is how many lanes we may send on — our lane count clamped to the
|
||||
// ports the peer bound — and so the modulus a flow's hash is reduced by.
|
||||
// Lanes from txLanes up can only receive, which is how a peer with more
|
||||
// routines than us still spreads its own traffic.
|
||||
// Immutable, and the length of every TX-side slice here.
|
||||
txLanes int
|
||||
|
||||
mu sync.Mutex
|
||||
|
||||
// peerPortCount and peerBasePort are the peer's advertised port range and
|
||||
// portOffset is this pair's rotation within it: lane s targets
|
||||
// peerBasePort + ((s + portOffset) % peerPortCount).
|
||||
peerPortCount uint16
|
||||
peerBasePort uint16
|
||||
portOffset uint16
|
||||
|
||||
// laneBias rotates the flow hash before it picks a lane, so the two sides of
|
||||
// a flow land on lanes that are each other's partner rather than at
|
||||
// independent points in the range. See newLaneSet.
|
||||
laneBias uint16
|
||||
|
||||
// peerAddr is the address the current lane targets were built from. The
|
||||
// peer's lane ports have no derivable relationship to a new NAT mapping, so
|
||||
// a roam invalidates every lane rather than moving it.
|
||||
peerAddr netip.Addr
|
||||
|
||||
// probe[s] is lane s's probe and backoff state. Sized txLanes.
|
||||
probe []laneProbeState
|
||||
}
|
||||
|
||||
// laneMaterial is everything a lane session is derived from. The two base keys
|
||||
// are the same secret the base tunnel's own cipher states already hold — the
|
||||
// noiseutil.CipherState interface just doesn't hand them back, so a lane set
|
||||
// keeps its own copy rather than a reference to the session.
|
||||
type laneMaterial struct {
|
||||
eKey, dKey [32]byte
|
||||
cipher noise.CipherFunc
|
||||
myCert cert.Certificate
|
||||
peerCert *cert.CachedCertificate
|
||||
initiator bool
|
||||
}
|
||||
|
||||
type laneProbeState struct {
|
||||
// gen is the generation of the last probe sent, echoed in the ack so a late
|
||||
// ack cannot promote a lane on the strength of a superseded probe.
|
||||
gen uint8
|
||||
|
||||
// fails is the consecutive failure count driving retryAt.
|
||||
fails uint8
|
||||
|
||||
// sentAt is when the outstanding probe went out, zero when none is pending.
|
||||
sentAt time.Time
|
||||
|
||||
// target is where the outstanding probe went, promoted to txAddr on ack. It
|
||||
// outlives the probe, which is what lets a demotion log the address that
|
||||
// stopped answering.
|
||||
target netip.AddrPort
|
||||
|
||||
// lastAck is when the lane was last confirmed usable, driving the keepalive.
|
||||
lastAck time.Time
|
||||
|
||||
// retryAt is the earliest we may probe this lane again.
|
||||
retryAt time.Time
|
||||
}
|
||||
|
||||
// newLaneSet sets up the lanes for a freshly completed base handshake. It
|
||||
// returns nil when the pair has no lane beyond the base tunnel, which is the
|
||||
// normal answer for a peer running without multiport. No session is derived
|
||||
// here; each is derived on the first packet that needs it.
|
||||
func newLaneSet(r *handshake.Result, myLanes int, myAddr, peerAddr netip.Addr) *laneSet {
|
||||
// PeerPortCount and PeerBasePort are already bounded to uint16 by the
|
||||
// handshake payload parser. A zero port count is a peer that did not
|
||||
// advertise multiport at all, so there is no lane to be had in either
|
||||
// direction.
|
||||
peerPorts := uint16(r.PeerPortCount)
|
||||
if peerPorts == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
// Sessions have to cover both directions: we send on our lanes and receive
|
||||
// on the peer's, and one derived session serves both ends of a lane index.
|
||||
// Only the session table is sized by the peer's advertised count; the TX-side
|
||||
// state is sized by what we will actually send on.
|
||||
n := min(max(myLanes, int(r.PeerTxLanes)), header.MaxLane+1)
|
||||
if n < 2 {
|
||||
return nil
|
||||
}
|
||||
txLanes := min(myLanes, int(peerPorts), n)
|
||||
|
||||
offset := lanePortOffset(myAddr, peerAddr, peerPorts)
|
||||
|
||||
// A flow picks its lane from a hash both sides compute identically, so with
|
||||
// the ranges lined up the two directions of a flow would pick the same lane
|
||||
// index — and lane s targets the peer's lane (s + portOffset), not lane s. The
|
||||
// high-addressed side rotates its choice by the low side's offset, which is
|
||||
// its own negated, so the two directions land on partner lanes and their
|
||||
// 4-tuples are exact reverses: each side's traffic then arrives through the
|
||||
// conntrack entry the other's probe opened. With mismatched ranges there are
|
||||
// no partner lanes to find, so don't pretend: hash straight.
|
||||
bias := uint16(0)
|
||||
if txLanes == int(peerPorts) && peerAddr.Less(myAddr) {
|
||||
bias = (peerPorts - offset) % peerPorts
|
||||
}
|
||||
|
||||
ls := &laneSet{
|
||||
sessions: make([]atomic.Pointer[ConnectionState], n),
|
||||
material: laneMaterial{
|
||||
eKey: r.EKey.UnsafeKey(),
|
||||
dKey: r.DKey.UnsafeKey(),
|
||||
cipher: r.Cipher,
|
||||
myCert: r.MyCert,
|
||||
peerCert: r.RemoteCert,
|
||||
initiator: r.Initiator,
|
||||
},
|
||||
txAddr: make([]atomic.Pointer[netip.AddrPort], txLanes),
|
||||
demand: make([]atomic.Bool, txLanes),
|
||||
probe: make([]laneProbeState, txLanes),
|
||||
txLanes: txLanes,
|
||||
peerPortCount: peerPorts,
|
||||
peerBasePort: uint16(r.PeerBasePort),
|
||||
portOffset: offset,
|
||||
laneBias: bias,
|
||||
}
|
||||
|
||||
// Every lane starts out demanded, so the first traffic tick on this tunnel
|
||||
// probes all of them at once instead of waiting for a flow to hash onto each.
|
||||
// Lanes have to be up *before* the flows are, not after: the peer writes a
|
||||
// flow's inbound packets to the tun queue matching the socket they arrived on,
|
||||
// and the kernel remembers that for as long as the flow stays busy. A flow that
|
||||
// starts while the lanes are still down therefore gets pinned to queue 0 on
|
||||
// both hosts for its whole life. This costs one probe per lane on any tunnel
|
||||
// with traffic; an idle tunnel is never ticked, so it still costs nothing.
|
||||
for s := 1; s < txLanes; s++ {
|
||||
ls.demand[s].Store(true)
|
||||
}
|
||||
return ls
|
||||
}
|
||||
|
||||
// laneLogAttr summarizes what a tunnel negotiated, for the handshake log lines.
|
||||
// It is an empty attr, which slog drops, on a node not running multiport. A peer
|
||||
// that negotiated no lanes still logs, with zeros: "we offered and got nothing"
|
||||
// is exactly what you want to see when you expected lanes and have none.
|
||||
func laneLogAttr(myLanes int, ls *laneSet) slog.Attr {
|
||||
if myLanes == 0 {
|
||||
return slog.Attr{}
|
||||
}
|
||||
if ls == nil {
|
||||
return slog.Any("lanes", m{"tx": 0, "sessions": 0})
|
||||
}
|
||||
// Every field read here is immutable once the set is built.
|
||||
return slog.Any("lanes", m{
|
||||
"tx": ls.txLanes,
|
||||
"sessions": len(ls.sessions),
|
||||
"peerBasePort": ls.peerBasePort,
|
||||
"peerPorts": ls.peerPortCount,
|
||||
"portOffset": ls.portOffset,
|
||||
})
|
||||
}
|
||||
|
||||
// emitLaneStats reports how many lanes are carrying traffic and how many tunnels
|
||||
// have any. Both are counted by walking the hostmap, because a counter kept at
|
||||
// promotion and demotion would drift upward forever: a tunnel torn down while its
|
||||
// lanes are up never demotes them. Only a node running multiport pays for the
|
||||
// walk.
|
||||
func (f *Interface) emitLaneStats(up, tunnels metrics.Gauge) {
|
||||
var nUp, nTunnels int64
|
||||
f.hostMap.ForEachIndex(func(hostinfo *HostInfo) {
|
||||
ls := hostinfo.lanes
|
||||
if ls == nil {
|
||||
return
|
||||
}
|
||||
nTunnels++
|
||||
for s := 1; s < ls.txLanes; s++ {
|
||||
if ls.txAddr[s].Load() != nil {
|
||||
nUp++
|
||||
}
|
||||
}
|
||||
})
|
||||
up.Update(nUp)
|
||||
tunnels.Update(nTunnels)
|
||||
}
|
||||
|
||||
// lanePortOffset returns the rotation applied to this pair's lane target ports,
|
||||
// in [0, peerPortCount). Without it every low-routine peer would aim its few
|
||||
// lanes at a big peer's first few ports, concentrating the big peer's receive
|
||||
// work on a couple of sockets; the hash spreads pairs across the whole range.
|
||||
//
|
||||
// Both sides hash the same sorted vpn-address pair and the higher address
|
||||
// negates the result, so when port counts match the two sides' rotations
|
||||
// cancel: our lane s's 4-tuple stays the reverse of the peer's lane s, and each
|
||||
// side's probe opens the conntrack entry the other's arrives through. (The one
|
||||
// lane a nonzero rotation lands on the peer's base port has no partner lane, so
|
||||
// behind a port-restricted NAT it is the one lane that may never come up.)
|
||||
func lanePortOffset(myAddr, peerAddr netip.Addr, peerPortCount uint16) uint16 {
|
||||
if peerPortCount == 0 {
|
||||
return 0
|
||||
}
|
||||
lo, hi := myAddr, peerAddr
|
||||
if hi.Less(lo) {
|
||||
lo, hi = hi, lo
|
||||
}
|
||||
h := fnv.New32a()
|
||||
b := lo.As16()
|
||||
h.Write(b[:])
|
||||
b = hi.As16()
|
||||
h.Write(b[:])
|
||||
o := uint16(h.Sum32() % uint32(peerPortCount))
|
||||
if myAddr == hi {
|
||||
o = (peerPortCount - o) % peerPortCount
|
||||
}
|
||||
return o
|
||||
}
|
||||
|
||||
// laneSession returns the session to decrypt a lane s packet with, deriving one
|
||||
// if this is the first packet to claim that lane. A nil session with no error
|
||||
// means this tunnel has no lane s at all.
|
||||
//
|
||||
// cached reports whether the session was already in the table. A fresh one is
|
||||
// deliberately left out of it: anyone who can spoof this tunnel's local index
|
||||
// can name any lane, and installing on sight would let them make us hold a
|
||||
// replay window and two cipher states per lane without authenticating anything.
|
||||
// The caller must install with installSession once the packet decrypts, which is
|
||||
// the first moment the lane is known to be real.
|
||||
func (i *HostInfo) laneSession(s uint8) (ci *ConnectionState, cached bool, err error) {
|
||||
ls := i.lanes
|
||||
if ls == nil || s == 0 || int(s) >= len(ls.sessions) {
|
||||
return nil, false, nil
|
||||
}
|
||||
|
||||
if cs := ls.sessions[s].Load(); cs != nil {
|
||||
return cs, true, nil
|
||||
}
|
||||
|
||||
cs, err := newLaneConnectionState(&ls.material, s)
|
||||
if err != nil {
|
||||
return nil, false, err
|
||||
}
|
||||
return cs, false, nil
|
||||
}
|
||||
|
||||
// installSession publishes a session derived by laneSession, so the next packet
|
||||
// on the lane doesn't have to derive it again. cs must have already decrypted the
|
||||
// packet at messageCounter.
|
||||
//
|
||||
// Two routines can race on a lane's first packet and derive a session each. The
|
||||
// loser's is dropped, and with it the replay-window entry for the packet it just
|
||||
// accepted, so hand that counter to the session that survives — the keys are
|
||||
// identical, so it is the same window in every respect that matters.
|
||||
func (ls *laneSet) installSession(l *slog.Logger, s uint8, cs *ConnectionState, messageCounter uint64) {
|
||||
if ls.sessions[s].CompareAndSwap(nil, cs) {
|
||||
return
|
||||
}
|
||||
ls.sessions[s].Load().noteSeen(l, messageCounter)
|
||||
}
|
||||
|
||||
// session returns lane s's session for our own use, deriving and installing it
|
||||
// if it doesn't exist yet. Unlike the RX path this needs no proof the lane is
|
||||
// real: we only ask for lanes we chose to send on. s must be a lane this set
|
||||
// covers.
|
||||
func (ls *laneSet) session(s int) (*ConnectionState, error) {
|
||||
if cs := ls.sessions[s].Load(); cs != nil {
|
||||
return cs, nil
|
||||
}
|
||||
|
||||
cs, err := newLaneConnectionState(&ls.material, uint8(s))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
if !ls.sessions[s].CompareAndSwap(nil, cs) {
|
||||
return ls.sessions[s].Load(), nil
|
||||
}
|
||||
return cs, nil
|
||||
}
|
||||
|
||||
// maxMessageCounter returns the highest counter across the base session and
|
||||
// every lane session. Data rides the lanes, so the base counter alone would
|
||||
// never reach the rehandshake or exhaustion thresholds and the lane keys would
|
||||
// be used past their data-volume margin. Rolling the base tunnel replaces the
|
||||
// lane keys with it, since lanes are derived from it.
|
||||
func (i *HostInfo) maxMessageCounter() uint64 {
|
||||
if i.ConnectionState == nil {
|
||||
return 0
|
||||
}
|
||||
c := i.ConnectionState.messageCounter.Load()
|
||||
if ls := i.lanes; ls != nil {
|
||||
for s := range ls.sessions {
|
||||
cs := ls.sessions[s].Load()
|
||||
if cs == nil {
|
||||
// Never derived, so it has never sent anything either.
|
||||
continue
|
||||
}
|
||||
if lc := cs.messageCounter.Load(); lc > c {
|
||||
c = lc
|
||||
}
|
||||
}
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
// txLane returns the session and destination for lane s, or a nil session when
|
||||
// the lane is down and the caller must use the base tunnel. A miss raises
|
||||
// demand, which is what gets a down lane probed again, so we pay for a lane
|
||||
// exactly where real traffic wanted one. Callers on the data plane come through
|
||||
// txLaneForFlow.
|
||||
func (ls *laneSet) txLane(s int) (*ConnectionState, netip.AddrPort) {
|
||||
if ls == nil || s <= 0 || s >= ls.txLanes {
|
||||
return nil, netip.AddrPort{}
|
||||
}
|
||||
|
||||
if addr := ls.txAddr[s].Load(); addr != nil {
|
||||
// The lane is only up because a probe was acked on it, and that probe
|
||||
// derived the session, so this load cannot miss. Fall back rather than
|
||||
// derive here anyway: this is the hot path and a nil is not worth an HKDF.
|
||||
if cs := ls.sessions[s].Load(); cs != nil {
|
||||
return cs, *addr
|
||||
}
|
||||
return nil, netip.AddrPort{}
|
||||
}
|
||||
|
||||
// Load-guarded so the common case of a lane that will not come up is a
|
||||
// plain read and cannot ping-pong the cache line these flags share.
|
||||
if !ls.demand[s].Load() {
|
||||
ls.demand[s].Store(true)
|
||||
}
|
||||
return nil, netip.AddrPort{}
|
||||
}
|
||||
|
||||
// txLaneForFlow picks the lane a flow rides and returns it with its session and
|
||||
// destination, or lane 0 and a nil session when the flow belongs on the base
|
||||
// tunnel — either because the hash landed on lane 0 or because the lane it
|
||||
// landed on is down.
|
||||
//
|
||||
// The lane comes from the flow rather than from the sending routine so that lane
|
||||
// use doesn't depend on how the kernel steers tun queues; see txQueue. It is a
|
||||
// pure function of the 5-tuple, so a flow stays on one lane for its life: no
|
||||
// per-packet reordering, and one lane's replay window sees one set of flows.
|
||||
func (ls *laneSet) txLaneForFlow(p *firewall.Packet) (int, *ConnectionState, netip.AddrPort) {
|
||||
if ls == nil || ls.txLanes < 2 {
|
||||
return 0, nil, netip.AddrPort{}
|
||||
}
|
||||
|
||||
s := int((laneFlowHash(p) + uint32(ls.laneBias)) % uint32(ls.txLanes))
|
||||
if s == 0 {
|
||||
// Lane 0 is the base tunnel, and a full share of flows belongs on it.
|
||||
return 0, nil, netip.AddrPort{}
|
||||
}
|
||||
|
||||
cs, addr := ls.txLane(s)
|
||||
return s, cs, addr
|
||||
}
|
||||
|
||||
// laneFlowHash hashes a 5-tuple to the same value from either end of the flow,
|
||||
// which is what lets both peers pick partner lanes for it (see laneBias). FNV-1a
|
||||
// by hand rather than through hash/fnv: this runs per packet, and the interface
|
||||
// there would escape the addresses to the heap.
|
||||
func laneFlowHash(p *firewall.Packet) uint32 {
|
||||
// Order the two endpoints so the direction of travel cannot change the hash.
|
||||
aAddr, aPort := p.LocalAddr, p.LocalPort
|
||||
bAddr, bPort := p.RemoteAddr, p.RemotePort
|
||||
if bAddr.Less(aAddr) || (aAddr == bAddr && bPort < aPort) {
|
||||
aAddr, aPort, bAddr, bPort = bAddr, bPort, aAddr, aPort
|
||||
}
|
||||
|
||||
const prime = 16777619
|
||||
h := uint32(2166136261)
|
||||
x, y := aAddr.As16(), bAddr.As16()
|
||||
for i := range x {
|
||||
h = (h ^ uint32(x[i])) * prime
|
||||
h = (h ^ uint32(y[i])) * prime
|
||||
}
|
||||
for _, b := range [5]byte{byte(aPort >> 8), byte(aPort), byte(bPort >> 8), byte(bPort), p.Protocol} {
|
||||
h = (h ^ uint32(b)) * prime
|
||||
}
|
||||
return h
|
||||
}
|
||||
|
||||
// laneTargetPortLocked returns the peer port lane s aims at. Only meaningful
|
||||
// when peerPortCount is nonzero, which txLanes > 0 guarantees.
|
||||
func (ls *laneSet) laneTargetPortLocked(s int) uint16 {
|
||||
return ls.peerBasePort + uint16((s+int(ls.portOffset))%int(ls.peerPortCount))
|
||||
}
|
||||
|
||||
// laneTargetLocked returns where lane s's probes go: the peer port this lane is
|
||||
// paired with, on the peer's current direct address.
|
||||
//
|
||||
// A lane only ever aims at its own port. There is no fallback to the peer's base
|
||||
// port when the lane port doesn't answer — a lane sharing the base port's
|
||||
// destination gains only a source port of its own, while costing the peer the
|
||||
// receive spread that is the whole point, so a lane that can't reach its port
|
||||
// stays down and its traffic rides the base tunnel.
|
||||
func (ls *laneSet) laneTargetLocked(s int, addr netip.Addr) netip.AddrPort {
|
||||
return netip.AddrPortFrom(addr, ls.laneTargetPortLocked(s))
|
||||
}
|
||||
|
||||
// laneRetryDelay is the backoff after fails consecutive probe failures.
|
||||
func laneRetryDelay(fails uint8) time.Duration {
|
||||
d := laneRetryBase << min(fails, 4)
|
||||
if d > laneRetryMax {
|
||||
d = laneRetryMax
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
// noteAck records an acked probe for lane s, promoting the lane if it was down.
|
||||
// gen must match the outstanding probe. Reports the target the lane came up on,
|
||||
// and whether this ack is what promoted it.
|
||||
func (ls *laneSet) noteAck(s int, gen uint8, now time.Time) (netip.AddrPort, bool) {
|
||||
if s <= 0 || s >= len(ls.probe) {
|
||||
return netip.AddrPort{}, false
|
||||
}
|
||||
|
||||
ls.mu.Lock()
|
||||
defer ls.mu.Unlock()
|
||||
|
||||
p := &ls.probe[s]
|
||||
if p.sentAt.IsZero() || p.gen != gen {
|
||||
// No probe outstanding, or an ack for a probe we have already given up
|
||||
// on. Either way it says nothing about the lane's current path.
|
||||
return netip.AddrPort{}, false
|
||||
}
|
||||
|
||||
p.sentAt = time.Time{}
|
||||
p.lastAck = now
|
||||
p.fails = 0
|
||||
p.retryAt = time.Time{}
|
||||
|
||||
target := p.target
|
||||
if ls.txAddr[s].Load() != nil {
|
||||
// Keepalive for a lane already up.
|
||||
return target, false
|
||||
}
|
||||
|
||||
ls.txAddr[s].Store(&target)
|
||||
return target, true
|
||||
}
|
||||
|
||||
// probeLanes runs one lane maintenance pass for a peer: it demotes lanes whose
|
||||
// probe went unanswered, re-proves lanes that have been up a while without one,
|
||||
// and probes down lanes the data plane asked for. Driven by the connection
|
||||
// manager's per-tunnel traffic tick, which only fires for a live tunnel — the
|
||||
// same condition that produces lane demand in the first place.
|
||||
func (f *Interface) probeLanes(hostinfo *HostInfo, now time.Time, nb, out []byte) {
|
||||
ls := hostinfo.lanes
|
||||
if ls == nil || ls.txLanes < 2 {
|
||||
return
|
||||
}
|
||||
|
||||
remote := hostinfo.GetRemote()
|
||||
|
||||
ls.mu.Lock()
|
||||
defer ls.mu.Unlock()
|
||||
|
||||
if !remote.IsValid() {
|
||||
// Relayed, or otherwise without a direct path. Lanes are direct-only,
|
||||
// so drop them all; a later tick rebuilds if a direct path returns.
|
||||
ls.resetLocked()
|
||||
return
|
||||
}
|
||||
|
||||
if ls.peerAddr != remote.Addr() {
|
||||
if ls.peerAddr.IsValid() {
|
||||
// A roam is a new path, not a failure: forget the lanes built on
|
||||
// the old one and let demand re-probe from a clean backoff. On the
|
||||
// first pass there is nothing built yet, so just record the address.
|
||||
ls.resetLocked()
|
||||
}
|
||||
ls.peerAddr = remote.Addr()
|
||||
}
|
||||
|
||||
for s := 1; s < ls.txLanes; s++ {
|
||||
p := &ls.probe[s]
|
||||
up := ls.txAddr[s].Load() != nil
|
||||
|
||||
if !p.sentAt.IsZero() {
|
||||
if now.Sub(p.sentAt) < laneProbeTimeout {
|
||||
continue
|
||||
}
|
||||
|
||||
// An aged-out probe is a failure whether it was bringing the lane
|
||||
// up or keeping it up.
|
||||
p.sentAt = time.Time{}
|
||||
p.fails = min(p.fails+1, laneMaxFails)
|
||||
p.retryAt = now.Add(laneRetryDelay(p.fails))
|
||||
if up {
|
||||
ls.txAddr[s].Store(nil)
|
||||
hostinfo.logger(f.l).Info("Multiport lane demoted, probe unanswered", "lane", s, "udpAddr", p.target)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
if up {
|
||||
if now.Sub(p.lastAck) < laneKeepalive {
|
||||
continue
|
||||
}
|
||||
} else if now.Before(p.retryAt) || !ls.demand[s].Swap(false) {
|
||||
continue
|
||||
}
|
||||
|
||||
p.gen++
|
||||
p.target = ls.laneTargetLocked(s, remote.Addr())
|
||||
if f.sendLaneProbe(hostinfo, s, p.gen, p.target, nb, out) {
|
||||
p.sentAt = now
|
||||
} else {
|
||||
p.fails = min(p.fails+1, laneMaxFails)
|
||||
p.retryAt = now.Add(laneRetryDelay(p.fails))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// resetLocked takes every lane down and clears its probe state, without
|
||||
// counting it as a failure.
|
||||
//
|
||||
// Demand is deliberately left standing: it records that a routine has real
|
||||
// traffic for this peer, which a roam or a relay detour does not change. Keeping
|
||||
// it re-probes the lanes that were actually carrying data as soon as a path
|
||||
// exists again, while a lane whose routine has gone quiet stays down.
|
||||
func (ls *laneSet) resetLocked() {
|
||||
for s := 1; s < len(ls.probe); s++ {
|
||||
ls.txAddr[s].Store(nil)
|
||||
ls.probe[s] = laneProbeState{}
|
||||
}
|
||||
}
|
||||
|
||||
// sendLaneProbe sends a probe on lane s to addr from a socket on lane s's port.
|
||||
// The probe is an ordinary Test packet encrypted with the lane's session, so an
|
||||
// ack proves the whole lane: our source port reached the peer, its reply reached
|
||||
// us, and the keys we derived for this lane match the ones it derived. Reports
|
||||
// whether the probe made it onto the wire.
|
||||
//
|
||||
// It goes out the first socket on the port, since the connection manager runs
|
||||
// this and has no queue of its own; every socket on the port has the same address,
|
||||
// so the probe proves the path for whichever one the data plane picks.
|
||||
func (f *Interface) sendLaneProbe(hostinfo *HostInfo, s int, gen uint8, addr netip.AddrPort, nb, out []byte) bool {
|
||||
// The first probe on a lane is what derives its session.
|
||||
ci, err := hostinfo.lanes.session(s)
|
||||
if err != nil {
|
||||
hostinfo.logger(f.l).Error("Failed to derive multiport lane session", "error", err, "lane", s)
|
||||
return false
|
||||
}
|
||||
if ci == nil || ci.eKey == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
ci.writeLock.Lock()
|
||||
}
|
||||
c, ok := ci.NextMessageCounter()
|
||||
if !ok {
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
ci.writeLock.Unlock()
|
||||
}
|
||||
f.dropExhausted(hostinfo, c, "Dropping multiport lane probe, lane message counter is exhausted")
|
||||
return false
|
||||
}
|
||||
|
||||
b := header.EncodeLane(out[:0], header.Version, header.Test, header.LaneProbe, hostinfo.remoteIndexId, c, uint8(s))
|
||||
b, err = ci.eKey.EncryptDanger(b, b, []byte{uint8(s), gen}, c, nb)
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
ci.writeLock.Unlock()
|
||||
}
|
||||
if err != nil {
|
||||
hostinfo.logger(f.l).Error("Failed to encrypt multiport lane probe", "error", err, "lane", s)
|
||||
return false
|
||||
}
|
||||
|
||||
f.messageMetrics.Tx(header.Test, header.LaneProbe, 1)
|
||||
if err := f.writers[f.laneSock(0, s)].WriteTo(b, addr); err != nil {
|
||||
hostinfo.logger(f.l).Error("Failed to send multiport lane probe", "error", err, "lane", s, "udpAddr", addr)
|
||||
return false
|
||||
}
|
||||
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
hostinfo.logger(f.l).Debug("Multiport lane probe sent", "lane", s, "gen", gen, "udpAddr", addr)
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// handleLaneProbe answers a peer's lane probe. The ack rides the base tunnel on
|
||||
// purpose: a probe proves the peer's lane s works in its send direction, and
|
||||
// answering on our own lane s would make the result depend on a second path
|
||||
// that may be broken independently.
|
||||
func (f *Interface) handleLaneProbe(hostinfo *HostInfo, lane uint8, payload []byte, rxc *rxContext) {
|
||||
if lane == 0 || len(payload) < 2 {
|
||||
return
|
||||
}
|
||||
|
||||
// Echo the header's lane rather than the payload's, so a peer cannot get us
|
||||
// to vouch for a lane it did not actually probe.
|
||||
f.send(header.Test, header.LaneProbeAck, hostinfo.ConnectionState, hostinfo,
|
||||
[]byte{lane, payload[1]}, rxc.nb, rxc.scratch[:0])
|
||||
}
|
||||
|
||||
// handleLaneProbeAck promotes the lane a peer just acked.
|
||||
func (f *Interface) handleLaneProbeAck(hostinfo *HostInfo, payload []byte) {
|
||||
ls := hostinfo.lanes
|
||||
if ls == nil || len(payload) < 2 {
|
||||
return
|
||||
}
|
||||
|
||||
// The target is worth logging next to the demotion that names the same
|
||||
// address, so a flapping lane can be read off the logs.
|
||||
if target, promoted := ls.noteAck(int(payload[0]), payload[1], time.Now()); promoted {
|
||||
hostinfo.logger(f.l).Info("Multiport lane up", "lane", payload[0], "udpAddr", target)
|
||||
}
|
||||
}
|
||||
+1035
File diff suppressed because it is too large
Load Diff
+1
-15
@@ -34,9 +34,7 @@ type LightHouse struct {
|
||||
|
||||
myVpnNetworks []netip.Prefix
|
||||
myVpnNetworksTable *bart.Lite
|
||||
// myVpnAddrsTable contains our overlay host addrs, as opposed to the overlay networks
|
||||
myVpnAddrsTable *bart.Lite
|
||||
punchy *Punchy
|
||||
punchy *Punchy
|
||||
|
||||
// localAddrsFn enumerates the underlay addresses we advertise. It is a field so tests can supply simulated
|
||||
// addresses rather than whatever this machine's NICs happen to be. Set it before Start.
|
||||
@@ -106,7 +104,6 @@ func NewLightHouseFromConfig(ctx context.Context, l *slog.Logger, c *config.C, c
|
||||
amLighthouse: amLighthouse,
|
||||
myVpnNetworks: cs.myVpnNetworks,
|
||||
myVpnNetworksTable: cs.myVpnNetworksTable,
|
||||
myVpnAddrsTable: cs.myVpnAddrsTable,
|
||||
addrMap: make(map[netip.Addr]*RemoteList),
|
||||
nebulaPort: nebulaPort,
|
||||
punchy: p,
|
||||
@@ -1161,17 +1158,6 @@ func (lhh *LightHouseHandler) handleHostQuery(n *NebulaMeta, fromVpnAddrs []neti
|
||||
return
|
||||
}
|
||||
|
||||
// Don't respond to requests for us.
|
||||
if lhh.lh.myVpnAddrsTable.Contains(queryVpnAddr) {
|
||||
if lhh.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
lhh.l.Debug("Ignoring HostQuery for one of my own addresses",
|
||||
"fromVpnAddrs", fromVpnAddrs,
|
||||
"queryVpnAddr", queryVpnAddr,
|
||||
)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
found, ln, err := lhh.lh.queryAndPrepMessage(queryVpnAddr, func(c *cache) (int, error) {
|
||||
n = lhh.resetMeta()
|
||||
n.Type = NebulaMeta_HostQueryReply
|
||||
|
||||
+53
-79
@@ -27,27 +27,15 @@ func TestOldIPv4Only(t *testing.T) {
|
||||
assert.Equal(t, binary.BigEndian.Uint32(bp[:]), m.GetAddr())
|
||||
}
|
||||
|
||||
func testCertState(networks ...netip.Prefix) *CertState {
|
||||
cs := &CertState{
|
||||
myVpnNetworks: networks,
|
||||
myVpnNetworksTable: new(bart.Lite),
|
||||
myVpnAddrs: make([]netip.Addr, 0, len(networks)),
|
||||
myVpnAddrsTable: new(bart.Lite),
|
||||
}
|
||||
|
||||
for _, n := range networks {
|
||||
cs.myVpnNetworksTable.Insert(n)
|
||||
cs.myVpnAddrs = append(cs.myVpnAddrs, n.Addr())
|
||||
cs.myVpnAddrsTable.Insert(netip.PrefixFrom(n.Addr(), n.Addr().BitLen()))
|
||||
}
|
||||
|
||||
return cs
|
||||
}
|
||||
|
||||
func Test_lhStaticMapping(t *testing.T) {
|
||||
l := test.NewLogger()
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/16")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
lh1 := "10.128.0.2"
|
||||
|
||||
c := config.NewC(l)
|
||||
@@ -67,7 +55,12 @@ func Test_lhStaticMapping(t *testing.T) {
|
||||
func TestReloadLighthouseInterval(t *testing.T) {
|
||||
l := test.NewLogger()
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/16")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
lh1 := "10.128.0.2"
|
||||
|
||||
c := config.NewC(l)
|
||||
@@ -97,7 +90,12 @@ func TestReloadLighthouseInterval(t *testing.T) {
|
||||
func BenchmarkLighthouseHandleRequest(b *testing.B) {
|
||||
l := test.NewLogger()
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/0")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
|
||||
c := config.NewC(l)
|
||||
lh, err := NewLightHouseFromConfig(b.Context(), l, c, cs, nil, nil)
|
||||
@@ -197,7 +195,12 @@ func TestLighthouse_Memory(t *testing.T) {
|
||||
c.Settings["listen"] = map[string]any{"port": 4242}
|
||||
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/24")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
lh, err := NewLightHouseFromConfig(t.Context(), l, c, cs, nil, nil)
|
||||
lh.ifce = &mockEncWriter{}
|
||||
require.NoError(t, err)
|
||||
@@ -277,7 +280,12 @@ func TestLighthouse_reload(t *testing.T) {
|
||||
c.Settings["listen"] = map[string]any{"port": 4242}
|
||||
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/24")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
|
||||
lh, err := NewLightHouseFromConfig(t.Context(), l, c, cs, nil, nil)
|
||||
require.NoError(t, err)
|
||||
@@ -307,7 +315,12 @@ func TestLighthouse_reloadStaticHostMap(t *testing.T) {
|
||||
}
|
||||
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/24")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
|
||||
lh, err := NewLightHouseFromConfig(t.Context(), l, c, cs, nil, nil)
|
||||
require.NoError(t, err)
|
||||
@@ -416,9 +429,7 @@ func TestLighthouse_reloadStaticHostMap(t *testing.T) {
|
||||
assert.Equal(t, []netip.AddrPort{netip.MustParseAddrPort("3.3.3.3:4242")}, rl.CopyAddrs([]netip.Prefix{}))
|
||||
}
|
||||
|
||||
// sendLHHostRequest delivers a HostQuery to lhh and hands back the writer that
|
||||
// captured what it emitted. Pass a nil filter to see every message.
|
||||
func sendLHHostRequest(fromAddr netip.AddrPort, myVpnIp, queryVpnIp netip.Addr, lhh *LightHouseHandler, filter *NebulaMeta_MessageType) *testEncWriter {
|
||||
func newLHHostRequest(fromAddr netip.AddrPort, myVpnIp, queryVpnIp netip.Addr, lhh *LightHouseHandler) testLhReply {
|
||||
req := &NebulaMeta{
|
||||
Type: NebulaMeta_HostQuery,
|
||||
Details: &NebulaMetaDetails{},
|
||||
@@ -436,59 +447,12 @@ func sendLHHostRequest(fromAddr netip.AddrPort, myVpnIp, queryVpnIp netip.Addr,
|
||||
panic(err)
|
||||
}
|
||||
|
||||
w := &testEncWriter{metaFilter: filter}
|
||||
lhh.HandleRequest(fromAddr, []netip.Addr{myVpnIp}, b, w)
|
||||
return w
|
||||
}
|
||||
|
||||
func newLHHostRequest(fromAddr netip.AddrPort, myVpnIp, queryVpnIp netip.Addr, lhh *LightHouseHandler) testLhReply {
|
||||
filter := NebulaMeta_HostQueryReply
|
||||
return sendLHHostRequest(fromAddr, myVpnIp, queryVpnIp, lhh, &filter).lastReply
|
||||
}
|
||||
|
||||
func TestLighthouse_IgnoresHostQueryForItself(t *testing.T) {
|
||||
// Validate that we don't answer host queries for our own address.
|
||||
l := test.NewLogger()
|
||||
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/24")
|
||||
myVpnIp := myVpnNet.Addr()
|
||||
|
||||
c := config.NewC(l)
|
||||
c.Settings["lighthouse"] = map[string]any{"am_lighthouse": true}
|
||||
c.Settings["listen"] = map[string]any{"port": 4242}
|
||||
// Add a static_host_map entry for ourselves, so our address
|
||||
// is in the addrMap.
|
||||
c.Settings["static_host_map"] = map[string]any{
|
||||
myVpnIp.String(): []any{"192.168.100.1:4242"},
|
||||
w := &testEncWriter{
|
||||
metaFilter: &filter,
|
||||
}
|
||||
|
||||
lh, err := NewLightHouseFromConfig(t.Context(), l, c, testCertState(myVpnNet), nil, nil)
|
||||
require.NoError(t, err)
|
||||
lh.ifce = &mockEncWriter{}
|
||||
lhh := lh.NewRequestHandler()
|
||||
|
||||
peerVpnIp := netip.MustParseAddr("10.128.0.2")
|
||||
peerUdpAddr := netip.MustParseAddrPort("10.0.0.2:4242")
|
||||
otherVpnIp := netip.MustParseAddr("10.128.0.3")
|
||||
otherUdpAddr := netip.MustParseAddrPort("10.0.0.3:4242")
|
||||
|
||||
newLHHostUpdate(peerUdpAddr, peerVpnIp, []netip.AddrPort{peerUdpAddr}, lhh)
|
||||
newLHHostUpdate(otherUdpAddr, otherVpnIp, []netip.AddrPort{otherUdpAddr}, lhh)
|
||||
|
||||
// Control: a query about a real peer is still answered, and still ends with
|
||||
// the punch notification aimed at the host that was asked about.
|
||||
w := sendLHHostRequest(peerUdpAddr, peerVpnIp, otherVpnIp, lhh, nil)
|
||||
require.NotNil(t, w.lastReply.msg)
|
||||
assert.Equal(t, NebulaMeta_HostPunchNotification, w.lastReply.msg.Type)
|
||||
assert.Equal(t, otherVpnIp, w.lastReply.vpnIp)
|
||||
|
||||
// Now validate that we don't send to ourselves.
|
||||
found, _, err := lh.queryAndPrepMessage(myVpnIp, func(*cache) (int, error) { return 0, nil })
|
||||
require.NoError(t, err)
|
||||
require.True(t, found, "the lighthouse should hold a cache entry for its own address")
|
||||
|
||||
w = sendLHHostRequest(peerUdpAddr, peerVpnIp, myVpnIp, lhh, nil)
|
||||
assert.Nil(t, w.lastReply.msg, "a query about our own address must produce no reply and no punch notification")
|
||||
lhh.HandleRequest(fromAddr, []netip.Addr{myVpnIp}, b, w)
|
||||
return w.lastReply
|
||||
}
|
||||
|
||||
func newLHHostUpdate(fromAddr netip.AddrPort, vpnIp netip.Addr, addrs []netip.AddrPort, lhh *LightHouseHandler) {
|
||||
@@ -678,7 +642,12 @@ func TestLighthouse_Dont_Delete_Static_Hosts(t *testing.T) {
|
||||
}
|
||||
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/24")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
lh, err := NewLightHouseFromConfig(t.Context(), l, c, cs, nil, nil)
|
||||
require.NoError(t, err)
|
||||
lh.ifce = &mockEncWriter{}
|
||||
@@ -739,7 +708,12 @@ func TestLighthouse_DeletesWork(t *testing.T) {
|
||||
}
|
||||
|
||||
myVpnNet := netip.MustParsePrefix("10.128.0.1/24")
|
||||
cs := testCertState(myVpnNet)
|
||||
nt := new(bart.Lite)
|
||||
nt.Insert(myVpnNet)
|
||||
cs := &CertState{
|
||||
myVpnNetworks: []netip.Prefix{myVpnNet},
|
||||
myVpnNetworksTable: nt,
|
||||
}
|
||||
lh, err := NewLightHouseFromConfig(t.Context(), l, c, cs, nil, nil)
|
||||
require.NoError(t, err)
|
||||
lh.ifce = &mockEncWriter{}
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"math"
|
||||
"net"
|
||||
"net/netip"
|
||||
"os"
|
||||
@@ -14,6 +15,7 @@ import (
|
||||
|
||||
"github.com/slackhq/nebula/config"
|
||||
"github.com/slackhq/nebula/cpupick"
|
||||
"github.com/slackhq/nebula/header"
|
||||
"github.com/slackhq/nebula/noiseutil"
|
||||
"github.com/slackhq/nebula/overlay"
|
||||
"github.com/slackhq/nebula/sshd"
|
||||
@@ -110,6 +112,105 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
l.Info("Using multiple routines", "routines", routines)
|
||||
}
|
||||
|
||||
port := c.GetInt("listen.port", 0)
|
||||
|
||||
batchSize := c.GetInt("listen.batch", 64)
|
||||
if batchSize < 1 {
|
||||
oldBatch := batchSize
|
||||
batchSize = 1
|
||||
l.Warn("listen.batch size is invalid", "provided", oldBatch, "overridden to", batchSize)
|
||||
}
|
||||
offloads := c.GetBool("listen.udp_offloads", false)
|
||||
|
||||
var listenHost netip.Addr
|
||||
if !configTest {
|
||||
rawListenHost := c.GetString("listen.host", "0.0.0.0")
|
||||
if rawListenHost == "[::]" {
|
||||
// Old guidance was to provide the literal `[::]` in `listen.host` but that won't resolve.
|
||||
listenHost = netip.IPv6Unspecified()
|
||||
|
||||
} else {
|
||||
ips, err := net.DefaultResolver.LookupNetIP(context.Background(), "ip", rawListenHost)
|
||||
if err != nil {
|
||||
return nil, util.ContextualizeIfNeeded("Failed to resolve listen.host", err)
|
||||
}
|
||||
if len(ips) == 0 {
|
||||
return nil, util.ContextualizeIfNeeded("Failed to resolve listen.host", err)
|
||||
}
|
||||
listenHost = ips[0].Unmap()
|
||||
}
|
||||
}
|
||||
|
||||
// Multiport lanes: bind a range of consecutive UDP ports (listen.port+p)
|
||||
// instead of SO_REUSEPORT-sharing one, and derive one extra session per lane
|
||||
// with capable peers, so a tunnel's inside flows spread over several underlay
|
||||
// 5-tuples instead of one. Defaults on, degrading gracefully when
|
||||
// preconditions aren't met — managed deployments (dnclient) can't be
|
||||
// hard-errored on config they don't control.
|
||||
//
|
||||
// `routines` is per port here rather than a total to divide up: every port
|
||||
// gets its own full set, and a port's routines share it through SO_REUSEPORT
|
||||
// so the kernel hashes each arriving 4-tuple onto one of them. That is what
|
||||
// keeps a port from being served by a single core — the base port above all,
|
||||
// since every handshake, every peer without multiport, and every tunnel whose
|
||||
// lanes are down or firewalled arrives there. A routine still owns exactly one
|
||||
// socket, which is what lets the read path own its state without locking, so
|
||||
// the worker count is routines * ports and everything sized by routines from
|
||||
// here down means that product.
|
||||
routinesPerPort := routines
|
||||
multiportPorts := 1
|
||||
multiport := c.GetBool("multiport.enabled", true)
|
||||
if multiport {
|
||||
multiportPorts = c.GetInt("multiport.ports", 0)
|
||||
if multiportPorts > header.MaxLane+1 {
|
||||
// A lane index rides in one byte of the nebula header, so a port we
|
||||
// could never address a lane on is a port we would never send from.
|
||||
l.Warn("multiport.ports clamped to the lane header limit", "ports", multiportPorts, "limit", header.MaxLane+1)
|
||||
multiportPorts = header.MaxLane + 1
|
||||
}
|
||||
if multiportPorts*routinesPerPort > maxRoutines {
|
||||
clamped := max(maxRoutines/routinesPerPort, 1)
|
||||
l.Warn("multiport.ports clamped to the routine limit",
|
||||
"ports", multiportPorts, "clampedTo", clamped, "routinesPerPort", routinesPerPort, "limit", maxRoutines)
|
||||
multiportPorts = clamped
|
||||
}
|
||||
if multiportPorts < 2 {
|
||||
// A single port is a plain SO_REUSEPORT listener, which is what a node
|
||||
// without multiport already runs. Say so rather than claiming lanes.
|
||||
l.Info("multiport disabled: set multiport.ports > 1 to bind a lane port range")
|
||||
multiport = false
|
||||
}
|
||||
}
|
||||
if multiport && port != 0 && port+multiportPorts-1 > math.MaxUint16 {
|
||||
l.Warn("multiport disabled: would bind ports beyond 65535", "listen.port", port, "ports", multiportPorts)
|
||||
multiport = false
|
||||
}
|
||||
if multiport && !configTest {
|
||||
// Every socket needs its own reader; a platform that can't run multiple
|
||||
// readers would silently strand all but one of them as blackholes. Probe
|
||||
// capability before sizing tun queues and routines to the port range.
|
||||
probe, err := udp.NewListener(l, udp.Settings{
|
||||
Listen: netip.AddrPortFrom(listenHost, 0),
|
||||
Batch: 1,
|
||||
})
|
||||
if err != nil {
|
||||
// We could not confirm support, so don't gamble a bound port range on it. The real bind below reports
|
||||
// the underlying error if it is not transient.
|
||||
l.Warn("multiport disabled: could not probe udp reader support", "error", err)
|
||||
multiport = false
|
||||
} else {
|
||||
if !probe.SupportsMultipleReaders() {
|
||||
l.Warn("multiport disabled: this platform does not support multiple udp readers")
|
||||
multiport = false
|
||||
}
|
||||
_ = probe.Close()
|
||||
}
|
||||
}
|
||||
if multiport {
|
||||
routines = routinesPerPort * multiportPorts
|
||||
l.Info("multiport routines", "ports", multiportPorts, "routinesPerPort", routinesPerPort, "routines", routines)
|
||||
}
|
||||
|
||||
// EXPERIMENTAL
|
||||
// Intentionally not documented yet while we do more testing and determine
|
||||
// a good default value.
|
||||
@@ -144,7 +245,6 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
|
||||
// set up our UDP listener
|
||||
udpConns := make([]udp.Conn, routines)
|
||||
port := c.GetInt("listen.port", 0)
|
||||
|
||||
// Callers get no handle to these until the Control is returned, release them on any error.
|
||||
defer func() {
|
||||
@@ -158,54 +258,72 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
}()
|
||||
|
||||
if !configTest {
|
||||
rawListenHost := c.GetString("listen.host", "0.0.0.0")
|
||||
var listenHost netip.Addr
|
||||
if rawListenHost == "[::]" {
|
||||
// Old guidance was to provide the literal `[::]` in `listen.host` but that won't resolve.
|
||||
listenHost = netip.IPv6Unspecified()
|
||||
|
||||
} else {
|
||||
ips, err := net.DefaultResolver.LookupNetIP(context.Background(), "ip", rawListenHost)
|
||||
if err != nil {
|
||||
return nil, util.ContextualizeIfNeeded("Failed to resolve listen.host", err)
|
||||
}
|
||||
if len(ips) == 0 {
|
||||
return nil, util.ContextualizeIfNeeded("Failed to resolve listen.host", err)
|
||||
}
|
||||
listenHost = ips[0].Unmap()
|
||||
}
|
||||
|
||||
for i := 0; i < routines; i++ {
|
||||
listen := netip.AddrPortFrom(listenHost, uint16(port))
|
||||
l.Info("listening", "addr", listen)
|
||||
batchSize := c.GetInt("listen.batch", 64)
|
||||
if batchSize < 1 {
|
||||
oldBatch := batchSize
|
||||
batchSize = 1
|
||||
l.Warn("listen.batch size is invalid", "provided", oldBatch, "overridden to", batchSize)
|
||||
}
|
||||
udpSettings := udp.Settings{
|
||||
Listen: listen,
|
||||
Multi: routines > 1,
|
||||
Batch: batchSize,
|
||||
Offloads: c.GetBool("listen.udp_offloads", false),
|
||||
}
|
||||
udpServer, err := udp.NewListener(l, udpSettings)
|
||||
if err != nil {
|
||||
return nil, util.NewContextualError("Failed to open udp listener", m{"queue": i}, err)
|
||||
}
|
||||
udpServer.ReloadConfig(c)
|
||||
udpConns[i] = udpServer
|
||||
|
||||
// If port is dynamic, discover it before the next pass through the for loop
|
||||
// This way all routines will use the same port correctly
|
||||
if port == 0 {
|
||||
uPort, err := udpServer.LocalAddr()
|
||||
if err != nil {
|
||||
return nil, util.NewContextualError("Failed to get listening port", nil, err)
|
||||
// With a dynamic listen.port, multiport binds the first socket dynamically
|
||||
// and then claims the next ports-1 above it; if that range turns out to be
|
||||
// partially occupied, re-roll with a fresh dynamic port.
|
||||
dynamic := port == 0
|
||||
var bindErr error
|
||||
for attempt := 0; attempt < 6; attempt++ {
|
||||
bindErr = nil
|
||||
for i := 0; i < routines; i++ {
|
||||
// Routines are laid out port-major: routine i serves port slot
|
||||
// i/routinesPerPort, so a port's routines are a contiguous run of
|
||||
// indices and cpu pinning, which walks the index, spreads each
|
||||
// port's group across the cores rather than stacking it on one.
|
||||
slot := 0
|
||||
if multiport {
|
||||
slot = i / routinesPerPort
|
||||
}
|
||||
port = int(uPort.Port())
|
||||
udpServer, err := udp.NewListener(l, udp.Settings{
|
||||
Listen: netip.AddrPortFrom(listenHost, uint16(port+slot)),
|
||||
// Under multiport the destination port narrows a packet to one
|
||||
// port's routines and SO_REUSEPORT picks among them by 4-tuple
|
||||
// hash, so both halves of the steering are load spreading: no
|
||||
// port depends on a single core, and a lane still has a source
|
||||
// port of its own to send from.
|
||||
Multi: routines > 1,
|
||||
Batch: batchSize,
|
||||
Offloads: offloads,
|
||||
})
|
||||
if err != nil {
|
||||
bindErr = util.NewContextualError("Failed to open udp listener", m{"queue": i}, err)
|
||||
break
|
||||
}
|
||||
udpServer.ReloadConfig(c)
|
||||
udpConns[i] = udpServer
|
||||
|
||||
// If port is dynamic, discover it before the next pass through the for loop
|
||||
// This way all routines will use the same port correctly
|
||||
if port == 0 {
|
||||
uPort, err := udpServer.LocalAddr()
|
||||
if err != nil {
|
||||
return nil, util.NewContextualError("Failed to get listening port", nil, err)
|
||||
}
|
||||
port = int(uPort.Port())
|
||||
if multiport && port+multiportPorts-1 > math.MaxUint16 {
|
||||
bindErr = util.NewContextualError("multiport dynamic port too close to 65535", m{"port": port}, nil)
|
||||
break
|
||||
}
|
||||
}
|
||||
l.Info("listening", "addr", netip.AddrPortFrom(listenHost, uint16(port+slot)), "socket", i)
|
||||
}
|
||||
if bindErr == nil {
|
||||
break
|
||||
}
|
||||
if !(multiport && dynamic) {
|
||||
return nil, bindErr
|
||||
}
|
||||
for i := range udpConns {
|
||||
if udpConns[i] != nil {
|
||||
_ = udpConns[i].Close()
|
||||
udpConns[i] = nil
|
||||
}
|
||||
}
|
||||
port = 0
|
||||
l.Debug("multiport dynamic port range collided, retrying", "attempt", attempt+1)
|
||||
}
|
||||
if bindErr != nil {
|
||||
return nil, bindErr
|
||||
}
|
||||
}
|
||||
|
||||
@@ -231,6 +349,26 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
messageMetrics: messageMetrics,
|
||||
}
|
||||
|
||||
if multiport {
|
||||
// A lane needs a port to send from, so we can send on no more lanes than we
|
||||
// bound ports. multiport.lanes below that sends on a subset of the range,
|
||||
// which is only interesting for narrowing an experiment: the ports are bound
|
||||
// and read either way.
|
||||
lanes := c.GetInt("multiport.lanes", 0)
|
||||
if lanes <= 0 || lanes > multiportPorts {
|
||||
lanes = multiportPorts
|
||||
}
|
||||
handshakeConfig.laneCount = lanes
|
||||
handshakeConfig.lanePortCount = uint16(multiportPorts)
|
||||
handshakeConfig.laneBasePort = uint16(port)
|
||||
|
||||
// Every other multiport log line is a reason it turned itself off, so say
|
||||
// plainly when it is on and with what. A peer only gets lanes if it also
|
||||
// advertises a port range, so this is our half of the negotiation.
|
||||
l.Info("multiport enabled", "lanes", lanes, "basePort", port, "ports", multiportPorts,
|
||||
"routinesPerPort", routinesPerPort)
|
||||
}
|
||||
|
||||
handshakeManager := NewHandshakeManager(l, hostMap, lightHouse, udpConns[0], handshakeConfig)
|
||||
lightHouse.handshakeTrigger = handshakeManager.trigger
|
||||
|
||||
@@ -286,6 +424,9 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
DropLocalBroadcast: c.GetBool("tun.drop_local_broadcast", false),
|
||||
DropMulticast: c.GetBool("tun.drop_multicast", false),
|
||||
routines: routines,
|
||||
Multiport: multiport,
|
||||
RoutinesPerPort: routinesPerPort,
|
||||
LaneCount: handshakeConfig.laneCount,
|
||||
MessageMetrics: messageMetrics,
|
||||
version: buildVersion,
|
||||
relayManager: NewRelayManager(ctx, l, hostMap, c),
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package noiseutil
|
||||
|
||||
import (
|
||||
"crypto/cipher"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
@@ -52,3 +53,29 @@ func NewCipherState(s *noise.CipherState, cipherFunc noise.CipherFunc) CipherSta
|
||||
panic(fmt.Sprintf("noiseutil: unsupported cipher %q", cipherFunc.CipherName()))
|
||||
}
|
||||
}
|
||||
|
||||
// NewCipherStateFromKey builds a data-plane CipherState directly from raw key
|
||||
// bytes, bypassing the noise handshake. Multiport lanes use this to key a
|
||||
// derived session off the base tunnel's negotiated keys; the caller owns the
|
||||
// guarantee that key is used with exactly one CipherState so nonces never
|
||||
// repeat.
|
||||
func NewCipherStateFromKey(key [32]byte, cipherFunc noise.CipherFunc) CipherState {
|
||||
c := cipherFunc.Cipher(key)
|
||||
if cs, ok := c.(CipherState); ok {
|
||||
return cs
|
||||
}
|
||||
|
||||
aead, ok := c.(cipher.AEAD)
|
||||
if !ok {
|
||||
panic(fmt.Sprintf("noiseutil: cipher %q does not expose an AEAD", cipherFunc.CipherName()))
|
||||
}
|
||||
|
||||
switch cipherFunc.CipherName() {
|
||||
case noise.CipherAESGCM.CipherName():
|
||||
return &CipherStateAESGCM{c: aead}
|
||||
case noise.CipherChaChaPoly.CipherName():
|
||||
return &CipherStateChaChaPoly{c: aead}
|
||||
default:
|
||||
panic(fmt.Sprintf("noiseutil: unsupported cipher %q", cipherFunc.CipherName()))
|
||||
}
|
||||
}
|
||||
|
||||
+70
-13
@@ -102,12 +102,49 @@ func (f *Interface) readOutsidePackets(via ViaSender, packet []byte, rxc *rxCont
|
||||
// recvError if necessary
|
||||
if hostinfo == nil || hostinfo.ConnectionState == nil {
|
||||
if !via.IsRelayed {
|
||||
f.maybeSendRecvError(via.UdpAddr, h.RemoteIndex)
|
||||
f.maybeSendRecvError(via.UdpAddr, h.RemoteIndex, via.SockIdx)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
if len(packet) < header.Len+hostinfo.ConnectionState.dKey.Overhead() {
|
||||
// Which session decrypts this packet is the lane index in the header. Lane 0
|
||||
// is the base tunnel; a higher lane is one of the sessions derived from it.
|
||||
ci := hostinfo.ConnectionState
|
||||
lane := h.Lane()
|
||||
laneCached := true
|
||||
if lane != 0 {
|
||||
if isMessageRelay {
|
||||
// A relay carrier is always the base tunnel, so lane ciphertext can
|
||||
// never legitimately arrive wrapped in one. Checked before the lookup
|
||||
// below so a junk relay packet can't make us derive a session.
|
||||
f.messageMetrics.RxInvalid(1)
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
hostinfo.logger(f.l).Debug("Refusing relayed multiport lane packet", "from", via, "header", h)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
var err error
|
||||
ci, laneCached, err = hostinfo.laneSession(lane)
|
||||
if err != nil {
|
||||
f.messageMetrics.RxInvalid(1)
|
||||
hostinfo.logger(f.l).Error("Failed to derive multiport lane session", "error", err, "lane", lane)
|
||||
return
|
||||
}
|
||||
if ci == nil {
|
||||
// A lane this tunnel doesn't have: a stale lane from a tunnel that has
|
||||
// since rolled, or a peer sending above what it advertised. Dropping
|
||||
// silently is right for both — a recv_error would tear down a
|
||||
// perfectly good base tunnel on the strength of one odd packet.
|
||||
f.messageMetrics.RxInvalid(1)
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
hostinfo.logger(f.l).Debug("Unknown multiport lane", "from", via, "header", h)
|
||||
}
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
if len(packet) < header.Len+ci.dKey.Overhead() {
|
||||
f.messageMetrics.RxInvalid(1)
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
f.l.Debug("packet too small", "from", via, "length", len(packet))
|
||||
@@ -118,7 +155,7 @@ func (f *Interface) readOutsidePackets(via ViaSender, packet []byte, rxc *rxCont
|
||||
// All remaining packets are encrypted
|
||||
if isMessageRelay {
|
||||
// Relay packets are special, this branch should always early-return
|
||||
err = hostinfo.ConnectionState.VerifyRelay(f.l, h.MessageCounter, packet, rxc.nb)
|
||||
err = ci.VerifyRelay(f.l, h.MessageCounter, packet, rxc.nb)
|
||||
if err != nil {
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
hostinfo.logger(f.l).Debug("Failed to verify relay packet", "error", err, "from", via, "header", h)
|
||||
@@ -129,7 +166,7 @@ func (f *Interface) readOutsidePackets(via ViaSender, packet []byte, rxc *rxCont
|
||||
return
|
||||
}
|
||||
|
||||
out, err := hostinfo.ConnectionState.Decrypt(f.l, h.MessageCounter, packet, rxc.nb)
|
||||
out, err := ci.Decrypt(f.l, h.MessageCounter, packet, rxc.nb)
|
||||
if err != nil {
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
hostinfo.logger(f.l).Debug("Failed to decrypt packet", "error", err, "from", via, "header", h)
|
||||
@@ -137,15 +174,25 @@ func (f *Interface) readOutsidePackets(via ViaSender, packet []byte, rxc *rxCont
|
||||
return
|
||||
}
|
||||
|
||||
// Roam before we respond
|
||||
f.handleHostRoaming(hostinfo, via)
|
||||
if !laneCached {
|
||||
// The packet decrypted, so the peer really is using this lane and the
|
||||
// session we derived for it is worth keeping.
|
||||
hostinfo.lanes.installSession(f.l, lane, ci, h.MessageCounter)
|
||||
}
|
||||
|
||||
// Roam before we respond, but only on the base tunnel: a lane's source
|
||||
// address is a per-lane 4-tuple, not the tunnel's remote, and letting it
|
||||
// roam the hostinfo would point every non-lane packet at a lane port.
|
||||
if lane == 0 {
|
||||
f.handleHostRoaming(hostinfo, via)
|
||||
}
|
||||
f.connectionManager.In(hostinfo)
|
||||
|
||||
switch h.Type {
|
||||
case header.Message:
|
||||
switch h.Subtype {
|
||||
case header.MessageNone:
|
||||
f.handleOutsideMessagePacket(hostinfo, h.MessageCounter, out, rxc)
|
||||
f.handleOutsideMessagePacket(hostinfo, ci, h.MessageCounter, out, rxc)
|
||||
default:
|
||||
hostinfo.logger(f.l).Error("IsValidSubType was true, but unexpected message subtype seen", "from", via, "header", h)
|
||||
return
|
||||
@@ -170,6 +217,10 @@ func (f *Interface) readOutsidePackets(via ViaSender, packet []byte, rxc *rxCont
|
||||
return
|
||||
}
|
||||
f.send(header.Test, header.TestReply, hostinfo.ConnectionState, hostinfo, out, rxc.nb, rxc.scratch[:0])
|
||||
case header.LaneProbe:
|
||||
f.handleLaneProbe(hostinfo, lane, out, rxc)
|
||||
case header.LaneProbeAck:
|
||||
f.handleLaneProbeAck(hostinfo, out)
|
||||
default:
|
||||
hostinfo.logger(f.l).Error("IsValidSubType was true, but unexpected test subtype seen", "from", via, "header", h)
|
||||
return
|
||||
@@ -214,6 +265,7 @@ func (f *Interface) handleOutsideRelayPacket(hostinfo *HostInfo, via ViaSender,
|
||||
relayHI: hostinfo,
|
||||
relay: relay,
|
||||
IsRelayed: true,
|
||||
SockIdx: via.SockIdx,
|
||||
}
|
||||
f.readOutsidePackets(via, signedPayload, rxc)
|
||||
case ForwardingType:
|
||||
@@ -470,7 +522,7 @@ func parseV4(data []byte, incoming bool, fp *firewall.ParsedPacket) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *Interface) handleOutsideMessagePacket(hostinfo *HostInfo, messageCounter uint64, out []byte, rxc *rxContext) {
|
||||
func (f *Interface) handleOutsideMessagePacket(hostinfo *HostInfo, ci *ConnectionState, messageCounter uint64, out []byte, rxc *rxContext) {
|
||||
err := newPacket(out, true, rxc.fwPacket)
|
||||
if err != nil {
|
||||
hostinfo.logger(f.l).Warn("Error while validating inbound packet", "error", err, "packet", out)
|
||||
@@ -479,6 +531,8 @@ func (f *Interface) handleOutsideMessagePacket(hostinfo *HostInfo, messageCounte
|
||||
|
||||
dropReason := f.firewall.Drop(rxc.fwPacket.Packet, true, hostinfo, f.pki.GetCAPool(), rxc.ctCache.Get())
|
||||
if dropReason != nil {
|
||||
// The reject rides the base tunnel: it is a control response, not lane
|
||||
// data, and the lane it arrived on says nothing about where it belongs.
|
||||
f.rejectOutside(out, hostinfo.ConnectionState, hostinfo, rxc.nb, rxc.scratch, rxc.q)
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
hostinfo.logger(f.l).Debug("dropping inbound packet", "fwPacket", rxc.fwPacket, "reason", dropReason)
|
||||
@@ -486,23 +540,26 @@ func (f *Interface) handleOutsideMessagePacket(hostinfo *HostInfo, messageCounte
|
||||
return
|
||||
}
|
||||
|
||||
err = f.batchers[rxc.q].Commit(out, batch.SortKey{Epoch: hostinfo.ConnectionState.epoch, Counter: messageCounter}, rxc.fwPacket)
|
||||
err = f.batchers[rxc.q].Commit(out, batch.SortKey{Epoch: ci.epoch, Counter: messageCounter}, rxc.fwPacket)
|
||||
if err != nil {
|
||||
f.l.Error("Failed to write to tun", "error", err)
|
||||
}
|
||||
}
|
||||
|
||||
func (f *Interface) maybeSendRecvError(endpoint netip.AddrPort, index uint32) {
|
||||
func (f *Interface) maybeSendRecvError(endpoint netip.AddrPort, index uint32, q int) {
|
||||
if f.sendRecvErrorConfig.ShouldRecvError(endpoint) {
|
||||
f.sendRecvError(endpoint, index)
|
||||
f.sendRecvError(endpoint, index, q)
|
||||
}
|
||||
}
|
||||
|
||||
func (f *Interface) sendRecvError(endpoint netip.AddrPort, index uint32) {
|
||||
// sendRecvError replies from the socket the offending packet arrived on (q).
|
||||
// A lane peer's spoof guard compares our source addr against the lane's
|
||||
// remote, so a reply from the base port would be discarded.
|
||||
func (f *Interface) sendRecvError(endpoint netip.AddrPort, index uint32, q int) {
|
||||
f.messageMetrics.Tx(header.RecvError, 0, 1)
|
||||
|
||||
b := header.Encode(make([]byte, header.Len), header.Version, header.RecvError, 0, index, 0)
|
||||
_ = f.outside.WriteTo(b, endpoint)
|
||||
_ = f.writers[q].WriteTo(b, endpoint)
|
||||
if f.l.Enabled(context.Background(), slog.LevelDebug) {
|
||||
f.l.Debug("Recv error sent",
|
||||
"index", index,
|
||||
|
||||
@@ -13,19 +13,36 @@ type batchWriter interface {
|
||||
// One SendBatch is owned by each listenIn goroutine; no locking is needed.
|
||||
// Slots are backed by an Arena (see its docs)
|
||||
type SendBatch struct {
|
||||
out batchWriter
|
||||
bufs [][]byte
|
||||
dsts []netip.AddrPort
|
||||
arena *Arena
|
||||
out batchWriter
|
||||
bufs [][]byte
|
||||
dsts []netip.AddrPort
|
||||
// arena backs the slots. borrowed means someone else owns it and Flush must
|
||||
// leave it alone.
|
||||
arena *Arena
|
||||
borrowed bool
|
||||
}
|
||||
|
||||
// NewSendBatch makes a SendBatch with batchCap slots and an arenaSize byte buffer for slices to back those slots
|
||||
func NewSendBatch(out batchWriter, batchCap, arenaSize int) *SendBatch {
|
||||
return newSendBatch(out, batchCap, NewArena(arenaSize), false)
|
||||
}
|
||||
|
||||
// NewSendBatchSharedArena makes a SendBatch that borrows arena rather than
|
||||
// allocating one. Several batches over different sockets can then split a single
|
||||
// slab, which is what multiport lanes need: the slab is the expensive part of a
|
||||
// SendBatch and one routine may hold a batch per lane. Flush leaves the arena
|
||||
// alone, so the owner must Reset it once every batch sharing it is drained.
|
||||
func NewSendBatchSharedArena(out batchWriter, batchCap int, arena *Arena) *SendBatch {
|
||||
return newSendBatch(out, batchCap, arena, true)
|
||||
}
|
||||
|
||||
func newSendBatch(out batchWriter, batchCap int, arena *Arena, borrowed bool) *SendBatch {
|
||||
return &SendBatch{
|
||||
out: out,
|
||||
bufs: make([][]byte, 0, batchCap),
|
||||
dsts: make([]netip.AddrPort, 0, batchCap),
|
||||
arena: NewArena(arenaSize),
|
||||
out: out,
|
||||
bufs: make([][]byte, 0, batchCap),
|
||||
dsts: make([]netip.AddrPort, 0, batchCap),
|
||||
arena: arena,
|
||||
borrowed: borrowed,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -54,6 +71,8 @@ func (b *SendBatch) Flush() (int, error) {
|
||||
clear(b.bufs)
|
||||
b.bufs = b.bufs[:0]
|
||||
b.dsts = b.dsts[:0]
|
||||
b.arena.Reset()
|
||||
if !b.borrowed {
|
||||
b.arena.Reset()
|
||||
}
|
||||
return written, err
|
||||
}
|
||||
|
||||
@@ -11,7 +11,6 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"slices"
|
||||
"sync/atomic"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
@@ -183,7 +182,6 @@ func (t *winTun) addRoutes(logErrors bool) error {
|
||||
luid := winipcfg.LUID(t.tun.LUID())
|
||||
routes := *t.Routes.Load()
|
||||
foundDefault4 := false
|
||||
carriesV6 := slices.ContainsFunc(t.vpnNetworks, func(p netip.Prefix) bool { return p.Addr().Is6() })
|
||||
|
||||
for _, r := range routes {
|
||||
if len(r.Via) == 0 || !r.Install {
|
||||
@@ -191,9 +189,6 @@ func (t *winTun) addRoutes(logErrors bool) error {
|
||||
continue
|
||||
}
|
||||
|
||||
// A v6 unsafe_route is legal under a v4-only cert; uninstalled ones put nothing on the adapter.
|
||||
carriesV6 = carriesV6 || r.Cidr.Addr().Is6()
|
||||
|
||||
// Add our unsafe route as an on-link route to the nebula tun device.
|
||||
err := luid.AddRoute(r.Cidr, unspecifiedNextHop(r.Cidr), uint32(r.Metric))
|
||||
if err != nil {
|
||||
@@ -215,11 +210,6 @@ func (t *winTun) addRoutes(logErrors bool) error {
|
||||
}
|
||||
}
|
||||
|
||||
return t.setMTU(luid, foundDefault4, carriesV6)
|
||||
}
|
||||
|
||||
// setMTU applies tun.mtu per address family. The default route metric rides along on the v4 handle.
|
||||
func (t *winTun) setMTU(luid winipcfg.LUID, foundDefault4, carriesV6 bool) error {
|
||||
ipif, err := luid.IPInterface(windows.AF_INET)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to get ip interface: %w", err)
|
||||
@@ -234,25 +224,6 @@ func (t *winTun) setMTU(luid winipcfg.LUID, foundDefault4, carriesV6 bool) error
|
||||
if err := ipif.Set(); err != nil {
|
||||
return fmt.Errorf("failed to set ip interface: %w", err)
|
||||
}
|
||||
|
||||
// Windows tracks NLMTU per family and wintun sets neither, so v6 keeps the adapter default of 65535.
|
||||
// Gated so a v4-only overlay under 1280 boots; a v6 one deliberately does not, as linux also refuses.
|
||||
if !carriesV6 {
|
||||
return nil
|
||||
}
|
||||
|
||||
ipif6, err := luid.IPInterface(windows.AF_INET6)
|
||||
if err != nil {
|
||||
// No v6 on the adapter means there is no NLMTU to get wrong. A failed Set below is not the same thing.
|
||||
t.l.Info("Skipping ipv6 MTU, no ipv6 interface on this adapter", "error", err)
|
||||
return nil
|
||||
}
|
||||
|
||||
ipif6.NLMTU = uint32(t.MTU)
|
||||
if err := ipif6.Set(); err != nil {
|
||||
return fmt.Errorf("failed to set ipv6 interface: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
"net/netip"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"unsafe"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
@@ -19,8 +20,14 @@ import (
|
||||
|
||||
// batchWriter owns the sendmmsg(2)/UDP-GSO transmit path for a StdConn: the
|
||||
// scratch WriteBatch packs mmsghdr entries into, plus the GSO capability
|
||||
// state probed at socket creation. Each queue has its own StdConn and
|
||||
// batchWriter, so no locking is needed.
|
||||
// state probed at socket creation.
|
||||
//
|
||||
// One socket can have several senders: under multiport every routine writes its
|
||||
// base-session traffic to socket 0, and any routine may write to any lane
|
||||
// socket, since a packet's lane comes from its flow rather than from the routine
|
||||
// that read it. The scratch is per socket, not per sender, so mu serializes
|
||||
// packing and draining. It is taken once per flush — a syscall's worth of work —
|
||||
// and only contends when two routines flush the same socket at the same instant.
|
||||
//
|
||||
// Terminology, smallest to largest:
|
||||
//
|
||||
@@ -42,6 +49,10 @@ type batchWriter struct {
|
||||
fd int
|
||||
isV4 bool
|
||||
|
||||
// mu guards everything below it: the scratch, and the GSO state WriteBatch
|
||||
// can clear.
|
||||
mu sync.Mutex
|
||||
|
||||
// UDP GSO (sendmsg with UDP_SEGMENT cmsg) support, probed once at
|
||||
// socket creation and cleared by WriteBatch if the kernel later rejects
|
||||
// a GSO send (the setsockopt probe cannot see per-route limitations).
|
||||
@@ -196,6 +207,9 @@ func (w *batchWriter) WriteBatch(bufs [][]byte, addrs []netip.AddrPort) (int, er
|
||||
return 0, fmt.Errorf("WriteBatch: len(bufs)=%d != len(addrs)=%d", len(bufs), len(addrs))
|
||||
}
|
||||
|
||||
w.mu.Lock()
|
||||
defer w.mu.Unlock()
|
||||
|
||||
// A destination the kernel rejects results in us dropping that entry (one packet, or one same-destination GSO run).
|
||||
// We count what actually made it out rather than returning an error.
|
||||
written := 0
|
||||
|
||||
+5
-18
@@ -31,8 +31,7 @@ func procyield(cycles uint32)
|
||||
|
||||
const (
|
||||
packetsPerRing = 1024
|
||||
// Caps tun.mtu at MTU-32 direct, MTU-64 relayed, unenforced anywhere else. 17.6MB page locked per socket.
|
||||
bytesPerPacket = MTU
|
||||
bytesPerPacket = 2048 - 32
|
||||
receiveSpins = 15
|
||||
)
|
||||
|
||||
@@ -70,14 +69,12 @@ func NewRIOListener(l *slog.Logger, addr netip.Addr, port int) (*RIOConn, error)
|
||||
|
||||
err := u.bind(l, &windows.SockaddrInet6{Addr: addr.As16(), Port: port})
|
||||
if err != nil {
|
||||
u.close()
|
||||
return nil, fmt.Errorf("bind: %w", err)
|
||||
}
|
||||
|
||||
for i := 0; i < packetsPerRing; i++ {
|
||||
err = u.insertReceiveRequest()
|
||||
if err != nil {
|
||||
u.close()
|
||||
return nil, fmt.Errorf("init rx ring: %w", err)
|
||||
}
|
||||
}
|
||||
@@ -359,25 +356,15 @@ func (u *RIOConn) Close() error {
|
||||
return nil
|
||||
}
|
||||
|
||||
u.close()
|
||||
return nil
|
||||
}
|
||||
|
||||
// Also unwinds a partial build from NewRIOListener, where isOpen is false and Close would no-op.
|
||||
// Socket first, unlike wireguard-go: receive() re-arms every slot, so freeing the rings under a live socket
|
||||
// hands the kernel freed pages for all packetsPerRing outstanding receives.
|
||||
func (u *RIOConn) close() {
|
||||
// WSASocket reports failure as InvalidHandle, not zero.
|
||||
if u.sock != 0 && u.sock != windows.InvalidHandle {
|
||||
windows.CloseHandle(u.sock)
|
||||
}
|
||||
u.sock = 0
|
||||
|
||||
windows.PostQueuedCompletionStatus(u.rx.iocp, 0, 0, nil)
|
||||
windows.PostQueuedCompletionStatus(u.tx.iocp, 0, 0, nil)
|
||||
|
||||
u.rx.CloseAndZero()
|
||||
u.tx.CloseAndZero()
|
||||
if u.sock != 0 {
|
||||
windows.CloseHandle(u.sock)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (ring *ringBuffer) Push() *ringPacket {
|
||||
|
||||
Reference in New Issue
Block a user