mirror of
https://github.com/slackhq/nebula.git
synced 2026-08-15 10:56:59 +02:00
datapath: fix 12 correctness findings from tun/UDP offload review
Multi-disciplinary correctness review of the batched tun / GSO-GRO / sendmmsg rework. Each fix has a regression test; the merged tree builds on linux/darwin/openbsd/windows/freebsd/netbsd, vets clean, passes the unit and e2e suites, and is -race clean. Critical: - C1 zero-length inner UDP datagram no longer panics the process (remote DoS): the UDP coalescer routes payLen==0 to passthrough instead of seeding a GSO slot, and WriteGSO skips empty payload iovecs as defense in depth. - C2 segmenter no longer corrupts inner headers when gsoSize < headerLen: the L3+L4 header is snapshotted once and each segment stamped from the copy, replacing the destructive overlapping in-place slide (SegmentTCP + SegmentUDP). High: - H1 applyOuterECN updates the IPv4 header checksum (RFC 1624 incremental) when folding outer CE into the inner ToS, so passthrough packets are no longer dropped by the peer stack. - H2 the GRO reject path caps the borrowed RX segment ([:n:n]) so a reject can no longer overrun into the next coalesced segment's Nebula header. Note: oversized ICMPv6 rejects that need >16B beyond the segment are now refused rather than sent under GRO (safe; see TOFIX.md for the scratch-buffer follow-up). - H3 WriteBatch falls back to per-packet WriteTo for a chunk when writeSockaddr fails, so one bad-family destination costs only its own packet, not the batch. - H4 UserDevice.Readers returns N distinct queue wrappers with private buffers (sharing the pipes) so concurrent readers no longer race/overwrite borrowed packet bytes. - H5 Poll.Close / Offload.Close no longer null t.fd (matching master's tunFile.Close), removing the data race with a concurrent readOne load. Medium/Low: - M1 the UDP GSO 127-segment gate moved from kernel >=5.5 to >=6.9 (the real UDP_MAX_SEGMENTS 64->128 threshold), avoiding EINVAL + per-packet fallback on 5.5-6.8 kernels. - M2 NewMultiQueueReader replays the offload mask newTun actually negotiated instead of the TSO-only mask, so adding a queue no longer disables USO device-wide; the advertised USO capability derives from the same mask. - M3 the shutdown eventfd is closed in pollQueueSet.Close / offloadQueueSet.Close (double-close guarded), fixing the per-lifecycle fd leak. - M4 dual-stack ECN selects the cmsg by address family, not socket family: RX parseRecvCmsg reads both IP_TOS and IPV6_TCLASS; TX writeEntryCmsg stamps IP_TOS for v4/v4-mapped dests and IPV6_TCLASS for v6 (on-host verified). - L1 newPoll no longer closes the fd on failure (matching newOffload), removing the double-close on QueueSet.Add error.
This commit is contained in:
@@ -1,8 +1,11 @@
|
||||
package nebula
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"log/slog"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/net/ipv4"
|
||||
)
|
||||
|
||||
func TestInnerECN(t *testing.T) {
|
||||
@@ -122,3 +125,64 @@ func TestApplyOuterECN(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyOuterECN_IPv4ChecksumStaysValid guards against H1: folding an outer
|
||||
// CE mark into the inner IPv4 ToS byte must keep the IPv4 header checksum valid.
|
||||
// The passthrough emit paths write the packet verbatim, so a stale checksum
|
||||
// turns an underlay congestion mark into packet loss at the receiver.
|
||||
func TestApplyOuterECN_IPv4ChecksumStaysValid(t *testing.T) {
|
||||
silent := slog.New(slog.DiscardHandler)
|
||||
hi := &HostInfo{}
|
||||
|
||||
// 20-byte IPv4 header with DSCP=0x88 and inner ECN = ECT(0). Folding CE
|
||||
// flips only the low two bits of the ToS byte while leaving DSCP intact.
|
||||
pkt := []byte{
|
||||
0x45, 0x88 | ecnECT0, 0, 40,
|
||||
0x1c, 0x46, 0x40, 0x00,
|
||||
64, 6, 0, 0,
|
||||
10, 0, 0, 1,
|
||||
10, 0, 0, 2,
|
||||
}
|
||||
// Stamp a correct header checksum before the fold.
|
||||
binary.BigEndian.PutUint16(pkt[10:12], ipv4HeaderChecksum(pkt[:ipv4.HeaderLen]))
|
||||
if !ipv4HeaderChecksumValid(pkt[:ipv4.HeaderLen]) {
|
||||
t.Fatal("test setup: initial header checksum invalid")
|
||||
}
|
||||
|
||||
applyOuterECN(pkt, ecnCE, hi, silent)
|
||||
|
||||
// CE folded in, DSCP preserved.
|
||||
if got, want := pkt[1], byte(0x88|ecnCE); got != want {
|
||||
t.Fatalf("ToS after fold = 0x%02x, want 0x%02x", got, want)
|
||||
}
|
||||
// The incremental RFC 1624 update must leave the checksum valid and equal
|
||||
// to a full recompute over the mutated header.
|
||||
if !ipv4HeaderChecksumValid(pkt[:ipv4.HeaderLen]) {
|
||||
t.Fatalf("IPv4 header checksum invalid after CE fold: 0x%04x", binary.BigEndian.Uint16(pkt[10:12]))
|
||||
}
|
||||
if got, want := binary.BigEndian.Uint16(pkt[10:12]), ipv4HeaderChecksum(pkt[:ipv4.HeaderLen]); got != want {
|
||||
t.Fatalf("checksum = 0x%04x, full recompute = 0x%04x", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// ipv4HeaderChecksum computes the RFC 1071 IPv4 header checksum over hdr,
|
||||
// treating the checksum field (bytes 10:12) as zero.
|
||||
func ipv4HeaderChecksum(hdr []byte) uint16 {
|
||||
var sum uint32
|
||||
for i := 0; i+1 < len(hdr); i += 2 {
|
||||
if i == 10 {
|
||||
continue // checksum field
|
||||
}
|
||||
sum += uint32(hdr[i])<<8 | uint32(hdr[i+1])
|
||||
}
|
||||
for sum > 0xffff {
|
||||
sum = (sum >> 16) + (sum & 0xffff)
|
||||
}
|
||||
return ^uint16(sum)
|
||||
}
|
||||
|
||||
// ipv4HeaderChecksumValid reports whether the stored checksum matches a fresh
|
||||
// computation over the header.
|
||||
func ipv4HeaderChecksumValid(hdr []byte) bool {
|
||||
return binary.BigEndian.Uint16(hdr[10:12]) == ipv4HeaderChecksum(hdr)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user