mirror of
https://github.com/slackhq/nebula.git
synced 2026-08-15 13:57:03 +02:00
datapath: fix 12 correctness findings from tun/UDP offload review
Multi-disciplinary correctness review of the batched tun / GSO-GRO / sendmmsg rework. Each fix has a regression test; the merged tree builds on linux/darwin/openbsd/windows/freebsd/netbsd, vets clean, passes the unit and e2e suites, and is -race clean. Critical: - C1 zero-length inner UDP datagram no longer panics the process (remote DoS): the UDP coalescer routes payLen==0 to passthrough instead of seeding a GSO slot, and WriteGSO skips empty payload iovecs as defense in depth. - C2 segmenter no longer corrupts inner headers when gsoSize < headerLen: the L3+L4 header is snapshotted once and each segment stamped from the copy, replacing the destructive overlapping in-place slide (SegmentTCP + SegmentUDP). High: - H1 applyOuterECN updates the IPv4 header checksum (RFC 1624 incremental) when folding outer CE into the inner ToS, so passthrough packets are no longer dropped by the peer stack. - H2 the GRO reject path caps the borrowed RX segment ([:n:n]) so a reject can no longer overrun into the next coalesced segment's Nebula header. Note: oversized ICMPv6 rejects that need >16B beyond the segment are now refused rather than sent under GRO (safe; see TOFIX.md for the scratch-buffer follow-up). - H3 WriteBatch falls back to per-packet WriteTo for a chunk when writeSockaddr fails, so one bad-family destination costs only its own packet, not the batch. - H4 UserDevice.Readers returns N distinct queue wrappers with private buffers (sharing the pipes) so concurrent readers no longer race/overwrite borrowed packet bytes. - H5 Poll.Close / Offload.Close no longer null t.fd (matching master's tunFile.Close), removing the data race with a concurrent readOne load. Medium/Low: - M1 the UDP GSO 127-segment gate moved from kernel >=5.5 to >=6.9 (the real UDP_MAX_SEGMENTS 64->128 threshold), avoiding EINVAL + per-packet fallback on 5.5-6.8 kernels. - M2 NewMultiQueueReader replays the offload mask newTun actually negotiated instead of the TSO-only mask, so adding a queue no longer disables USO device-wide; the advertised USO capability derives from the same mask. - M3 the shutdown eventfd is closed in pollQueueSet.Close / offloadQueueSet.Close (double-close guarded), fixing the per-lifecycle fd leak. - M4 dual-stack ECN selects the cmsg by address family, not socket family: RX parseRecvCmsg reads both IP_TOS and IPV6_TCLASS; TX writeEntryCmsg stamps IP_TOS for v4/v4-mapped dests and IPV6_TCLASS for v6 (on-host verified). - L1 newPoll no longer closes the fd on failure (matching newOffload), removing the double-close on QueueSet.Add error.
This commit is contained in:
@@ -1,6 +1,7 @@
|
||||
package iputil
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"net"
|
||||
"testing"
|
||||
@@ -179,6 +180,51 @@ func Test_CreateRejectPacket_NoICMPError(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Test_CreateRejectPacket_RespectsCap guards against H2: with UDP GRO the
|
||||
// scratch buffer reused to build a reject is a single coalesced segment inside
|
||||
// a shared recvmmsg row. Its length covers just that segment, but an uncapped
|
||||
// slice's capacity runs on into the next, not-yet-processed segment. Because
|
||||
// CreateRejectPacket honors cap, capping the borrowed segment to its own length
|
||||
// (cap==len) makes it physically impossible for an oversized ICMPv6 reject to
|
||||
// overwrite the neighbor segment's bytes.
|
||||
func Test_CreateRejectPacket_RespectsCap(t *testing.T) {
|
||||
src := net.ParseIP("fd00::1")
|
||||
dst := net.ParseIP("fd00::2")
|
||||
|
||||
// Inner IPv6 UDP packet. An ICMPv6 reject copies the whole inner packet
|
||||
// plus a 48-byte header (40 IPv6 + 8 ICMPv6), so it needs 48 more bytes
|
||||
// than the inner packet length.
|
||||
inner := makeIPv6Packet(src, dst, 17, make([]byte, 20))
|
||||
|
||||
// The ciphertext scratch reused as the reject buffer is the received
|
||||
// datagram: 16-byte Nebula header + inner + 16-byte AEAD tag. That is only
|
||||
// 32 bytes of slack, so a full ICMPv6 reject overruns it by 16 bytes.
|
||||
const nebulaOverhead = 32
|
||||
segLen := len(inner) + nebulaOverhead
|
||||
|
||||
// Shared backing row laid out as [segment][neighbor's 16-byte Nebula header].
|
||||
const neighborHdr = 16
|
||||
sentinel := bytes.Repeat([]byte{0xAB}, neighborHdr)
|
||||
|
||||
// Uncapped: the slice's capacity reaches into the neighbor, reproducing
|
||||
// the overrun that silently drops the neighbor packet.
|
||||
backing := make([]byte, segLen+neighborHdr)
|
||||
copy(backing[segLen:], sentinel)
|
||||
reject := CreateRejectPacket(inner, backing[:segLen])
|
||||
assert.NotNil(t, reject, "uncapped buffer reaches into the neighbor, so the reject is built")
|
||||
assert.NotEqual(t, sentinel, backing[segLen:segLen+neighborHdr],
|
||||
"without the cap the oversized reject overruns into the neighbor segment")
|
||||
|
||||
// Capped (the fix): cap==len, so the builder cannot exceed the segment. The
|
||||
// reject does not fit, so it is refused rather than corrupting the neighbor.
|
||||
backing = make([]byte, segLen+neighborHdr)
|
||||
copy(backing[segLen:], sentinel)
|
||||
reject = CreateRejectPacket(inner, backing[:segLen:segLen])
|
||||
assert.Nil(t, reject, "capped segment is 16 bytes too small for a full ICMPv6 reject, so it is refused")
|
||||
assert.Equal(t, sentinel, backing[segLen:segLen+neighborHdr],
|
||||
"capped segment must leave the neighbor untouched")
|
||||
}
|
||||
|
||||
func makeIPv6Packet(src, dst net.IP, nextHeader uint8, payload []byte) []byte {
|
||||
b := make([]byte, ipv6.HeaderLen+len(payload))
|
||||
b[0] = ipv6.Version << 4
|
||||
|
||||
Reference in New Issue
Block a user