Files
nebula/main.go
T
Wade Simmons 53c565eb29 Pick multiport lanes by inside flow hash, not routine index
Lanes were collapsing onto lane 0 and staying there. Nebula writes a
peer's inbound packets to the tun queue matching the socket they arrived
on, and tun_flow_update teaches the kernel to steer that flow's outbound
packets to the same queue, preferring what it learned over the hash. So
while a tunnel's lanes were still down -- which is every tunnel, for its
first moments -- all traffic arrived on socket 0, pinning every flow to
queue 0 on both hosts for as long as it stayed busy. With the lane
following the routine, that meant lane 0 forever.

Two changes break the loop:

Choose the lane from the packet's own 5-tuple hash, so lane spread no
longer depends on tun steering at all. The hash is symmetric, and the
high-addressed side rotates its choice by the low side's port offset, so
a flow's two directions land on partner lanes with exact reverse
4-tuples and each arrives through the conntrack entry the other's probe
opened.

Seed every lane as demanded at lane-set creation, so the first traffic
tick probes them all at once rather than waiting for a flow to hash onto
each. Lanes have to be up before the flows are, not after.

A routine may now send on any lane, so it holds a SendBatch per lane,
built lazily and all borrowing one shared arena -- the slab is the
expensive part, and a batch per (routine, lane) would otherwise cost
hundreds of megabytes. full() counts across the batches so the total
outstanding stays bounded as before.

That also means several routines can write to one socket, which was
already true for base traffic under multiport: the linux batchWriter's
sendmmsg scratch had no lock. Take a mutex there, once per flush.
2026-09-03 14:24:53 -04:00

531 lines
17 KiB
Go

package nebula
import (
"context"
"fmt"
"log/slog"
"math"
"net"
"net/netip"
"os"
"runtime/debug"
"slices"
"strings"
"time"
"github.com/slackhq/nebula/config"
"github.com/slackhq/nebula/cpupick"
"github.com/slackhq/nebula/header"
"github.com/slackhq/nebula/noiseutil"
"github.com/slackhq/nebula/overlay"
"github.com/slackhq/nebula/sshd"
"github.com/slackhq/nebula/udp"
"github.com/slackhq/nebula/util"
"go.yaml.in/yaml/v3"
)
type m = map[string]any
// maxRoutines caps routines below the RejectHeadroom nonce gap so concurrent senders can't race the counter past wrap.
const maxRoutines = 1 << 16
// The reject headroom must exceed every sender that can be mid-reservation at once, about two per routine.
const _ = noiseutil.RejectHeadroom - 4*maxRoutines
func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, deviceFactory overlay.DeviceFactory) (retcon *Control, reterr error) {
ctx, cancel := context.WithCancel(context.Background())
// Automatically cancel the context if Main returns an error, to signal all created goroutines to quit.
defer func() {
if reterr != nil {
cancel()
}
}()
if buildVersion == "" {
buildVersion = moduleVersion()
}
// Debug builds (-tags debug) serve pprof on :6060; a no-op otherwise.
startPprofServer(ctx, l)
// Print the config if in test, the exit comes later
if configTest {
b, err := yaml.Marshal(c.Settings)
if err != nil {
return nil, err
}
// Print the final config
l.Info(string(b))
}
pki, err := NewPKIFromConfig(l, c)
if err != nil {
return nil, util.ContextualizeIfNeeded("Failed to load PKI from config", err)
}
fw, err := NewFirewallFromConfig(l, pki.getCertState(), c)
if err != nil {
return nil, util.ContextualizeIfNeeded("Error while loading firewall rules", err)
}
l.Info("Firewall started", "firewallHashes", fw.GetRuleHashes())
ssh, err := sshd.NewSSHServer(ctx, l.With("subsystem", "sshd"))
if err != nil {
return nil, util.ContextualizeIfNeeded("Error while creating SSH server", err)
}
wireSSHReload(l, ssh, c)
var sshStart func()
if c.GetBool("sshd.enabled", false) {
sshStart, err = configSSH(l, ssh, c)
if err != nil {
l.Warn("Failed to configure sshd, ssh debugging will not be available", "error", err)
sshStart = nil
}
}
////////////////////////////////////////////////////////////////////////////////////////////////////////////////////
// All non system modifying configuration consumption should live above this line
// tun config, listeners, anything modifying the computer should be below
////////////////////////////////////////////////////////////////////////////////////////////////////////////////////
var routines int
// If `routines` is set, use that and ignore the specific values
if routines = c.GetInt("routines", 0); routines != 0 {
if routines < 1 {
routines = 1
}
} else {
// deprecated and undocumented
tunQueues := c.GetInt("tun.routines", 1)
udpQueues := c.GetInt("listen.routines", 1)
routines = max(tunQueues, udpQueues)
if routines != 1 {
l.Warn("Setting tun.routines and listen.routines is deprecated. Use `routines` instead", "routines", routines)
}
}
if routines > maxRoutines {
l.Warn("Using multiple routines", "routines", maxRoutines, "clamped", true, "requestedRoutines", routines)
routines = maxRoutines
} else if routines > 1 {
l.Info("Using multiple routines", "routines", routines)
}
// EXPERIMENTAL
// Intentionally not documented yet while we do more testing and determine
// a good default value.
conntrackCacheTimeout := c.GetDuration("firewall.conntrack.routine_cache_timeout", 0)
if routines > 1 && !c.IsSet("firewall.conntrack.routine_cache_timeout") {
// Use a different default if we are running with multiple routines
conntrackCacheTimeout = 1 * time.Second
}
if conntrackCacheTimeout > 0 {
l.Info("Using routine-local conntrack cache", "duration", conntrackCacheTimeout)
}
var tun overlay.Device
if !configTest {
c.CatchHUP(ctx)
if deviceFactory == nil {
deviceFactory = overlay.NewDeviceFromConfig
}
tun, err = deviceFactory(c, l, pki.getCertState().myVpnNetworks, routines)
if err != nil {
return nil, util.ContextualizeIfNeeded("Failed to get a tun/tap device", err)
}
defer func() {
if reterr != nil {
tun.Close()
}
}()
}
// set up our UDP listener
udpConns := make([]udp.Conn, routines)
port := c.GetInt("listen.port", 0)
// Multiport lanes: bind `routines` consecutive UDP ports (listen.port+i)
// instead of SO_REUSEPORT sharing one, and derive one extra session per lane
// with capable peers so a tunnel's inside flows spread over several underlay
// 5-tuples instead of one. Defaults on, degrading gracefully when preconditions
// aren't met — managed deployments (dnclient) can't be hard-errored on
// config they don't control.
multiport := c.GetBool("multiport.enabled", true)
if multiport && routines < 2 {
l.Info("multiport disabled: requires routines > 1")
multiport = false
}
if multiport && port != 0 && port+routines-1 > math.MaxUint16 {
l.Warn("multiport disabled: would bind ports beyond 65535", "listen.port", port, "routines", routines)
multiport = false
}
// Callers get no handle to these until the Control is returned, release them on any error.
defer func() {
if reterr != nil {
for _, u := range udpConns {
if u != nil {
_ = u.Close()
}
}
}
}()
if !configTest {
rawListenHost := c.GetString("listen.host", "0.0.0.0")
var listenHost netip.Addr
if rawListenHost == "[::]" {
// Old guidance was to provide the literal `[::]` in `listen.host` but that won't resolve.
listenHost = netip.IPv6Unspecified()
} else {
ips, err := net.DefaultResolver.LookupNetIP(context.Background(), "ip", rawListenHost)
if err != nil {
return nil, util.ContextualizeIfNeeded("Failed to resolve listen.host", err)
}
if len(ips) == 0 {
return nil, util.ContextualizeIfNeeded("Failed to resolve listen.host", err)
}
listenHost = ips[0].Unmap()
}
batchSize := c.GetInt("listen.batch", 64)
if batchSize < 1 {
oldBatch := batchSize
batchSize = 1
l.Warn("listen.batch size is invalid", "provided", oldBatch, "overridden to", batchSize)
}
offloads := c.GetBool("listen.udp_offloads", false)
// Every lane socket needs its own reader; a platform that can't run
// multiple readers would silently strand sockets 1..n-1 as blackholes.
// Probe capability before committing to per-port binds.
if multiport {
probe, err := udp.NewListener(l, udp.Settings{
Listen: netip.AddrPortFrom(listenHost, 0),
Batch: 1,
})
if err != nil {
// We could not confirm support, so don't gamble a bound port range on it. The real bind below reports
// the underlying error if it is not transient.
l.Warn("multiport disabled: could not probe udp reader support", "error", err)
multiport = false
} else {
if !probe.SupportsMultipleReaders() {
l.Warn("multiport disabled: this platform does not support multiple udp readers")
multiport = false
}
_ = probe.Close()
}
}
// With a dynamic listen.port, multiport binds socket 0 dynamically and
// then claims the next routines-1 ports above it; if that range turns
// out to be partially occupied, re-roll with a fresh dynamic port.
dynamic := port == 0
var bindErr error
for attempt := 0; attempt < 6; attempt++ {
bindErr = nil
for i := 0; i < routines; i++ {
lPort := port
if multiport {
lPort = port + i
}
udpServer, err := udp.NewListener(l, udp.Settings{
Listen: netip.AddrPortFrom(listenHost, uint16(lPort)),
// Multiport gives every routine its own port, so SO_REUSEPORT sharing is neither needed nor wanted:
// the destination port alone must decide which socket, and therefore which routine, sees a flow.
Multi: routines > 1 && !multiport,
Batch: batchSize,
Offloads: offloads,
})
if err != nil {
bindErr = util.NewContextualError("Failed to open udp listener", m{"queue": i}, err)
break
}
udpServer.ReloadConfig(c)
udpConns[i] = udpServer
// If port is dynamic, discover it before the next pass through the for loop
// This way all routines will use the same port correctly
if port == 0 {
uPort, err := udpServer.LocalAddr()
if err != nil {
return nil, util.NewContextualError("Failed to get listening port", nil, err)
}
port = int(uPort.Port())
if multiport && port+routines-1 > math.MaxUint16 {
bindErr = util.NewContextualError("multiport dynamic port too close to 65535", m{"port": port}, nil)
break
}
}
bound := port
if multiport {
bound = port + i
}
l.Info("listening", "addr", netip.AddrPortFrom(listenHost, uint16(bound)), "socket", i)
}
if bindErr == nil {
break
}
if !(multiport && dynamic) {
return nil, bindErr
}
for i := range udpConns {
if udpConns[i] != nil {
_ = udpConns[i].Close()
udpConns[i] = nil
}
}
port = 0
l.Debug("multiport dynamic port range collided, retrying", "attempt", attempt+1)
}
if bindErr != nil {
return nil, bindErr
}
}
hostMap := NewHostMapFromConfig(l, c)
punchy := NewPunchyFromConfig(l, c, udpConns[0])
connManager := newConnectionManagerFromConfig(l, c, hostMap, punchy)
lightHouse, err := NewLightHouseFromConfig(ctx, l, c, pki.getCertState(), udpConns[0], punchy)
if err != nil {
return nil, util.ContextualizeIfNeeded("Failed to initialize lighthouse handler", err)
}
var messageMetrics *MessageMetrics
if c.GetBool("stats.message_metrics", false) {
messageMetrics = newMessageMetrics()
} else {
messageMetrics = newMessageMetricsOnlyRecvError()
}
handshakeConfig := HandshakeConfig{
tryInterval: c.GetDuration("handshakes.try_interval", DefaultHandshakeTryInterval),
retries: int64(c.GetInt("handshakes.retries", DefaultHandshakeRetries)),
triggerBuffer: c.GetInt("handshakes.trigger_buffer", DefaultHandshakeTriggerBuffer),
messageMetrics: messageMetrics,
}
if multiport {
lanes := c.GetInt("multiport.lanes", 0)
if lanes <= 0 || lanes > routines {
lanes = routines
}
if lanes > header.MaxLane+1 {
// The lane index rides in one byte of the nebula header, so lanes
// above that are unaddressable.
l.Warn("multiport.lanes clamped to header limit", "lanes", lanes, "limit", header.MaxLane+1)
lanes = header.MaxLane + 1
}
handshakeConfig.laneCount = lanes
handshakeConfig.lanePortCount = uint16(routines)
handshakeConfig.laneBasePort = uint16(port)
// Every other multiport log line is a reason it turned itself off, so say
// plainly when it is on and with what. A peer only gets lanes if it also
// advertises a port range, so this is our half of the negotiation.
l.Info("multiport enabled", "lanes", lanes, "basePort", port, "ports", routines)
}
handshakeManager := NewHandshakeManager(l, hostMap, lightHouse, udpConns[0], handshakeConfig)
lightHouse.handshakeTrigger = handshakeManager.trigger
ds, err := newDnsServerFromConfig(ctx, l, pki, hostMap, c)
if err != nil {
l.Warn("Failed to start DNS responder", "error", err)
}
pinThreads := c.GetBool("tun.pin_threads", true)
cpuAffinity := parseCpuAffinity(c, l, routines)
if pinThreads && routines > 1 && len(cpuAffinity) == 0 && !configTest {
// The operator didn't choose pin CPUs, so pick a default set that
// prefers performance cores and doesn't stack co-located instances
// onto allowed[0].
// key is used to seed the spreading of routines->cores.
// use PID if you want to ensure many different Nebulas in VMs or containers land on different cores
// use port if you want to always end up on the same cores, ideal for benchmarking.
key := uint64(os.Getpid()) //default to PID
pinKeyStr := strings.ToLower(c.GetString("tun.pin_threads_key", ""))
switch pinKeyStr {
case "":
l.Debug("tun.pin_threads_key is empty, using PID")
case "pid":
l.Debug("tun.pin_threads_key is PID")
case "port":
if ap, err := udpConns[0].LocalAddr(); err == nil && ap.Port() != 0 {
l.Info("tun.pin_threads_key is port number")
key = uint64(ap.Port())
} else {
l.Warn("Failed to get a port number for tun.pin_threads_key, falling back to PID", "err", err)
}
default:
l.Warn("tun.pin_threads_key is invalid, using PID")
}
cpuAffinity = cpupick.Default(routines, key, l)
}
ifConfig := &InterfaceConfig{
HostMap: hostMap,
Inside: tun,
Outside: udpConns[0],
pki: pki,
Firewall: fw,
DnsServer: ds,
HandshakeManager: handshakeManager,
connectionManager: connManager,
lightHouse: lightHouse,
tryPromoteEvery: c.GetUint32("counters.try_promote", defaultPromoteEvery),
reQueryEvery: c.GetUint32("counters.requery_every_packets", defaultReQueryEvery),
reQueryWait: c.GetDuration("timers.requery_wait_duration", defaultReQueryWait),
DropLocalBroadcast: c.GetBool("tun.drop_local_broadcast", false),
DropMulticast: c.GetBool("tun.drop_multicast", false),
routines: routines,
Multiport: multiport,
LaneCount: handshakeConfig.laneCount,
MessageMetrics: messageMetrics,
version: buildVersion,
relayManager: NewRelayManager(ctx, l, hostMap, c),
punchy: punchy,
ConntrackCacheTimeout: conntrackCacheTimeout,
CpuAffinity: cpuAffinity,
PinThreads: pinThreads,
l: l,
}
var ifce *Interface
if !configTest {
ifce, err = NewInterface(ctx, ifConfig)
if err != nil {
return nil, fmt.Errorf("failed to initialize interface: %s", err)
}
ifce.writers = udpConns
lightHouse.ifce = ifce
ifce.RegisterConfigChangeCallbacks(c)
ifce.reloadDisconnectInvalid(c)
ifce.reloadSendRecvError(c)
ifce.reloadAcceptRecvError(c)
handshakeManager.f = ifce
go handshakeManager.Run(ctx)
punchy.Start(ctx, ifce, hostMap, lightHouse)
}
stats, err := newStatsServerFromConfig(ctx, l, c, buildVersion, configTest)
if err != nil {
return nil, util.ContextualizeIfNeeded("Failed to start stats emitter", err)
}
if configTest {
return nil, nil
}
go ifce.emitStats(ctx, c.GetDuration("stats.interval", time.Second*10))
attachCommands(l, c, ssh, ifce)
networkChanges := udp.NewNetworkChangeMonitor(ctx, l, c)
return &Control{
state: StateReady,
f: ifce,
l: l,
ctx: ctx,
cancel: cancel,
sshStart: sshStart,
statsStart: stats.Start,
dnsStart: ds.Start,
lighthouseStart: lightHouse.StartUpdateWorker,
networkChangeStart: networkChanges.Start,
connectionManagerStart: connManager.Start,
}, nil
}
// parseCpuAffinity reads `tun.cpu_affinity` from the config — a list of
// integer CPU IDs, one per TUN reader goroutine. Empty / unset returns nil
// (listenIn falls back to spreading queues across the allowed CPU set).
// Length mismatch with `routines` is a warning, not an error: shorter lists
// are modulo-cycled across queues, longer lists' tail is ignored. Invalid
// entries (non-integer, or a CPU ID we're not allowed to run on) are also a
// warning and disable the override entirely so we don't silently pin to the
// wrong CPU. Entries are validated against the process's current affinity
// mask (util.AllowedCPUs) rather than 0..NumCPU-1: under a cgroup cpuset or
// taskset the runnable IDs are frequently not that contiguous range, and
// pinning to an unrunnable ID always fails. If the allowed set can't be
// determined we fall back to a plain non-negative check.
func parseCpuAffinity(c *config.C, l *slog.Logger, routines int) []int {
raw := c.Get("tun.cpu_affinity")
if raw == nil {
return nil
}
rv, ok := raw.([]any)
if !ok {
l.Warn("tun.cpu_affinity must be a list of integers; ignoring", "value", raw)
return nil
}
// allowed is the set of CPU IDs we're actually permitted to run on. A nil
// slice (unsupported platform or lookup error) means "can't tell", so we
// only apply the weaker non-negative check in that case.
allowed, err := util.AllowedCPUs()
if err != nil {
l.Warn("could not determine allowed CPUs; validating tun.cpu_affinity against non-negative only", "error", err)
allowed = nil
}
cpus := make([]int, 0, len(rv))
for i, e := range rv {
var cpu int
switch v := e.(type) {
case int:
cpu = v
case int64:
cpu = int(v)
case float64:
cpu = int(v)
default:
l.Warn("tun.cpu_affinity entry not an integer; ignoring affinity",
"index", i, "value", e)
return nil
}
if cpu < 0 {
l.Warn("tun.cpu_affinity entry out of range; ignoring affinity",
"index", i, "cpu", cpu)
return nil
}
if len(allowed) > 0 && !slices.Contains(allowed, cpu) {
l.Warn("tun.cpu_affinity entry not in allowed CPU set; ignoring affinity",
"index", i, "cpu", cpu, "allowed", allowed)
return nil
}
cpus = append(cpus, cpu)
}
if len(cpus) != routines {
l.Warn("tun.cpu_affinity length doesn't match routines; queues will modulo-cycle through the list",
"affinity_len", len(cpus), "routines", routines)
}
return cpus
}
func moduleVersion() string {
info, ok := debug.ReadBuildInfo()
if !ok {
return ""
}
for _, dep := range info.Deps {
if dep.Path == "github.com/slackhq/nebula" {
return strings.TrimPrefix(dep.Version, "v")
}
}
return ""
}