mirror of
https://github.com/slackhq/nebula.git
synced 2026-08-15 18:17:02 +02:00
44dd2e9ca4
Multi-disciplinary correctness review of the batched tun / GSO-GRO / sendmmsg rework. Each fix has a regression test; the merged tree builds on linux/darwin/openbsd/windows/freebsd/netbsd, vets clean, passes the unit and e2e suites, and is -race clean. Critical: - C1 zero-length inner UDP datagram no longer panics the process (remote DoS): the UDP coalescer routes payLen==0 to passthrough instead of seeding a GSO slot, and WriteGSO skips empty payload iovecs as defense in depth. - C2 segmenter no longer corrupts inner headers when gsoSize < headerLen: the L3+L4 header is snapshotted once and each segment stamped from the copy, replacing the destructive overlapping in-place slide (SegmentTCP + SegmentUDP). High: - H1 applyOuterECN updates the IPv4 header checksum (RFC 1624 incremental) when folding outer CE into the inner ToS, so passthrough packets are no longer dropped by the peer stack. - H2 the GRO reject path caps the borrowed RX segment ([:n:n]) so a reject can no longer overrun into the next coalesced segment's Nebula header. Note: oversized ICMPv6 rejects that need >16B beyond the segment are now refused rather than sent under GRO (safe; see TOFIX.md for the scratch-buffer follow-up). - H3 WriteBatch falls back to per-packet WriteTo for a chunk when writeSockaddr fails, so one bad-family destination costs only its own packet, not the batch. - H4 UserDevice.Readers returns N distinct queue wrappers with private buffers (sharing the pipes) so concurrent readers no longer race/overwrite borrowed packet bytes. - H5 Poll.Close / Offload.Close no longer null t.fd (matching master's tunFile.Close), removing the data race with a concurrent readOne load. Medium/Low: - M1 the UDP GSO 127-segment gate moved from kernel >=5.5 to >=6.9 (the real UDP_MAX_SEGMENTS 64->128 threshold), avoiding EINVAL + per-packet fallback on 5.5-6.8 kernels. - M2 NewMultiQueueReader replays the offload mask newTun actually negotiated instead of the TSO-only mask, so adding a queue no longer disables USO device-wide; the advertised USO capability derives from the same mask. - M3 the shutdown eventfd is closed in pollQueueSet.Close / offloadQueueSet.Close (double-close guarded), fixing the per-lifecycle fd leak. - M4 dual-stack ECN selects the cmsg by address family, not socket family: RX parseRecvCmsg reads both IP_TOS and IPV6_TCLASS; TX writeEntryCmsg stamps IP_TOS for v4/v4-mapped dests and IPV6_TCLASS for v6 (on-host verified). - L1 newPoll no longer closes the fd on failure (matching newOffload), removing the double-close on QueueSet.Add error.
169 lines
3.8 KiB
Go
169 lines
3.8 KiB
Go
//go:build linux && !android
|
|
// +build linux,!android
|
|
|
|
package tio
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
"sync/atomic"
|
|
|
|
"golang.org/x/sys/unix"
|
|
)
|
|
|
|
// Maximum size we accept for a single read from a TUN with IFF_VNET_HDR. A
|
|
// TSO superpacket can be up to 64KiB of payload plus a single L2/L3/L4 header
|
|
// prefix plus the virtio header.
|
|
const tunReadBufSize = 65535
|
|
|
|
type Poll struct {
|
|
fd int
|
|
|
|
readPoll [2]unix.PollFd
|
|
writePoll [2]unix.PollFd
|
|
closed atomic.Bool
|
|
|
|
readBuf []byte
|
|
batchRet [1]Packet
|
|
}
|
|
|
|
// newPoll wraps an existing tun fd. On failure it does NOT close fd: the
|
|
// caller owns fd and is the sole closer (see pollQueueSet.Add callers in
|
|
// overlay/tun_linux.go, which unix.Close on Add error). This matches the
|
|
// newOffload convention and keeps closes at exactly one on every path.
|
|
func newPoll(fd int, shutdownFd int) (*Poll, error) {
|
|
if err := unix.SetNonblock(fd, true); err != nil {
|
|
return nil, fmt.Errorf("failed to set Poll device as nonblocking: %w", err)
|
|
}
|
|
|
|
out := &Poll{
|
|
fd: fd,
|
|
readBuf: make([]byte, tunReadBufSize),
|
|
readPoll: [2]unix.PollFd{
|
|
{Fd: int32(fd), Events: unix.POLLIN},
|
|
{Fd: int32(shutdownFd), Events: unix.POLLIN},
|
|
},
|
|
writePoll: [2]unix.PollFd{
|
|
{Fd: int32(fd), Events: unix.POLLOUT},
|
|
{Fd: int32(shutdownFd), Events: unix.POLLIN},
|
|
},
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// blockOnRead waits until the Poll fd is readable or shutdown has been signaled.
|
|
// Returns os.ErrClosed if Close was called.
|
|
func (t *Poll) blockOnRead() error {
|
|
const problemFlags = unix.POLLHUP | unix.POLLNVAL | unix.POLLERR
|
|
var err error
|
|
for {
|
|
_, err = unix.Poll(t.readPoll[:], -1)
|
|
if err != unix.EINTR {
|
|
break
|
|
}
|
|
}
|
|
tunEvents := t.readPoll[0].Revents
|
|
shutdownEvents := t.readPoll[1].Revents
|
|
t.readPoll[0].Revents = 0
|
|
t.readPoll[1].Revents = 0
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if shutdownEvents&(unix.POLLIN|problemFlags) != 0 {
|
|
return os.ErrClosed
|
|
}
|
|
if tunEvents&problemFlags != 0 {
|
|
return os.ErrClosed
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func (t *Poll) blockOnWrite() error {
|
|
const problemFlags = unix.POLLHUP | unix.POLLNVAL | unix.POLLERR
|
|
var err error
|
|
for {
|
|
_, err = unix.Poll(t.writePoll[:], -1)
|
|
if err != unix.EINTR {
|
|
break
|
|
}
|
|
}
|
|
tunEvents := t.writePoll[0].Revents
|
|
shutdownEvents := t.writePoll[1].Revents
|
|
t.writePoll[0].Revents = 0
|
|
t.writePoll[1].Revents = 0
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if shutdownEvents&(unix.POLLIN|problemFlags) != 0 {
|
|
return os.ErrClosed
|
|
}
|
|
if tunEvents&problemFlags != 0 {
|
|
return os.ErrClosed
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func (t *Poll) Read() ([]Packet, error) {
|
|
n, err := t.readOne(t.readBuf)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
t.batchRet[0] = Packet{Bytes: t.readBuf[:n]}
|
|
return t.batchRet[:], nil
|
|
}
|
|
|
|
func (t *Poll) readOne(to []byte) (int, error) {
|
|
for {
|
|
n, errno := unix.Read(t.fd, to)
|
|
if errno == nil {
|
|
return n, nil
|
|
}
|
|
switch errno {
|
|
case unix.EAGAIN:
|
|
if err := t.blockOnRead(); err != nil {
|
|
return 0, err
|
|
}
|
|
case unix.EINTR:
|
|
// retry
|
|
case unix.EBADF:
|
|
return 0, os.ErrClosed
|
|
default:
|
|
return 0, errno
|
|
}
|
|
}
|
|
}
|
|
|
|
// Write is only valid for single threaded use
|
|
func (t *Poll) Write(from []byte) (int, error) {
|
|
for {
|
|
n, errno := unix.Write(t.fd, from)
|
|
if errno == nil {
|
|
return n, nil
|
|
}
|
|
switch errno {
|
|
case unix.EAGAIN:
|
|
if err := t.blockOnWrite(); err != nil {
|
|
return 0, err
|
|
}
|
|
case unix.EINTR:
|
|
// retry
|
|
case unix.EBADF:
|
|
return 0, os.ErrClosed
|
|
default:
|
|
return 0, errno
|
|
}
|
|
}
|
|
}
|
|
|
|
func (t *Poll) Close() error {
|
|
if t.closed.Swap(true) {
|
|
return nil
|
|
}
|
|
//shutdownFd is owned by the container, so we should not close it
|
|
// Close the underlying fd but do NOT null t.fd: a reader may still be
|
|
// loading it in readOne, and mutating the field would race that load.
|
|
// It gets EBADF -> os.ErrClosed (or wakes via the shutdown eventfd's
|
|
// ppoll first). closed.Swap already guarantees we only close once.
|
|
return unix.Close(t.fd)
|
|
}
|