mirror of
https://github.com/slackhq/nebula.git
synced 2026-08-15 15:37:03 +02:00
44dd2e9ca4
Multi-disciplinary correctness review of the batched tun / GSO-GRO / sendmmsg rework. Each fix has a regression test; the merged tree builds on linux/darwin/openbsd/windows/freebsd/netbsd, vets clean, passes the unit and e2e suites, and is -race clean. Critical: - C1 zero-length inner UDP datagram no longer panics the process (remote DoS): the UDP coalescer routes payLen==0 to passthrough instead of seeding a GSO slot, and WriteGSO skips empty payload iovecs as defense in depth. - C2 segmenter no longer corrupts inner headers when gsoSize < headerLen: the L3+L4 header is snapshotted once and each segment stamped from the copy, replacing the destructive overlapping in-place slide (SegmentTCP + SegmentUDP). High: - H1 applyOuterECN updates the IPv4 header checksum (RFC 1624 incremental) when folding outer CE into the inner ToS, so passthrough packets are no longer dropped by the peer stack. - H2 the GRO reject path caps the borrowed RX segment ([:n:n]) so a reject can no longer overrun into the next coalesced segment's Nebula header. Note: oversized ICMPv6 rejects that need >16B beyond the segment are now refused rather than sent under GRO (safe; see TOFIX.md for the scratch-buffer follow-up). - H3 WriteBatch falls back to per-packet WriteTo for a chunk when writeSockaddr fails, so one bad-family destination costs only its own packet, not the batch. - H4 UserDevice.Readers returns N distinct queue wrappers with private buffers (sharing the pipes) so concurrent readers no longer race/overwrite borrowed packet bytes. - H5 Poll.Close / Offload.Close no longer null t.fd (matching master's tunFile.Close), removing the data race with a concurrent readOne load. Medium/Low: - M1 the UDP GSO 127-segment gate moved from kernel >=5.5 to >=6.9 (the real UDP_MAX_SEGMENTS 64->128 threshold), avoiding EINVAL + per-packet fallback on 5.5-6.8 kernels. - M2 NewMultiQueueReader replays the offload mask newTun actually negotiated instead of the TSO-only mask, so adding a queue no longer disables USO device-wide; the advertised USO capability derives from the same mask. - M3 the shutdown eventfd is closed in pollQueueSet.Close / offloadQueueSet.Close (double-close guarded), fixing the per-lifecycle fd leak. - M4 dual-stack ECN selects the cmsg by address family, not socket family: RX parseRecvCmsg reads both IP_TOS and IPV6_TCLASS; TX writeEntryCmsg stamps IP_TOS for v4/v4-mapped dests and IPV6_TCLASS for v6 (on-host verified). - L1 newPoll no longer closes the fd on failure (matching newOffload), removing the double-close on QueueSet.Add error.
91 lines
1.8 KiB
Go
91 lines
1.8 KiB
Go
//go:build linux && !android
|
|
// +build linux,!android
|
|
|
|
package tio
|
|
|
|
import (
|
|
"encoding/binary"
|
|
"errors"
|
|
"fmt"
|
|
"sync/atomic"
|
|
|
|
"golang.org/x/sys/unix"
|
|
)
|
|
|
|
type pollQueueSet struct {
|
|
pq []*Poll
|
|
// pqi is exactly the same as pq, but stored as the interface type
|
|
pqi []Queue
|
|
shutdownFd int
|
|
closed atomic.Bool
|
|
}
|
|
|
|
func NewPollQueueSet() (QueueSet, error) {
|
|
shutdownFd, err := unix.Eventfd(0, unix.EFD_NONBLOCK|unix.EFD_CLOEXEC)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("failed to create eventfd: %w", err)
|
|
}
|
|
|
|
out := &pollQueueSet{
|
|
pq: []*Poll{},
|
|
pqi: []Queue{},
|
|
shutdownFd: shutdownFd,
|
|
}
|
|
|
|
return out, nil
|
|
}
|
|
|
|
func (c *pollQueueSet) Queues() []Queue {
|
|
return c.pqi
|
|
}
|
|
|
|
func (c *pollQueueSet) Add(fd int) error {
|
|
x, err := newPoll(fd, c.shutdownFd)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
c.pq = append(c.pq, x)
|
|
c.pqi = append(c.pqi, x)
|
|
|
|
return nil
|
|
}
|
|
|
|
func (c *pollQueueSet) wakeForShutdown() error {
|
|
var buf [8]byte
|
|
binary.NativeEndian.PutUint64(buf[:], 1)
|
|
_, err := unix.Write(int(c.shutdownFd), buf[:])
|
|
return err
|
|
}
|
|
|
|
func (c *pollQueueSet) Close() error {
|
|
if c.closed.Swap(true) {
|
|
return nil
|
|
}
|
|
|
|
errs := []error{}
|
|
|
|
// Wake any reader blocked in poll so it observes POLLIN on the shutdown
|
|
// eventfd and returns os.ErrClosed.
|
|
if err := c.wakeForShutdown(); err != nil {
|
|
errs = append(errs, err)
|
|
}
|
|
|
|
// Close the per-queue tun fds; this also unblocks any in-flight reads.
|
|
// The per-queue Close deliberately leaves shutdownFd alone - it belongs
|
|
// to this container.
|
|
for _, x := range c.pq {
|
|
if err := x.Close(); err != nil {
|
|
errs = append(errs, err)
|
|
}
|
|
}
|
|
|
|
// Close the shutdown eventfd last: every reader's pollfd set references
|
|
// it, so it must outlive the wake + per-queue teardown above.
|
|
if err := unix.Close(c.shutdownFd); err != nil {
|
|
errs = append(errs, err)
|
|
}
|
|
c.shutdownFd = -1
|
|
|
|
return errors.Join(errs...)
|
|
}
|