mirror of
https://github.com/slackhq/nebula.git
synced 2026-08-16 02:56:59 +02:00
stuff
This commit is contained in:
+6
-8
@@ -14,14 +14,12 @@ import (
|
|||||||
|
|
||||||
const ReplayWindow = 8192
|
const ReplayWindow = 8192
|
||||||
|
|
||||||
// sessionEpoch hands out a receiver-local ordinal to every ConnectionState as
|
// sessionEpoch hands out a receiver-local ordinal to every ConnectionState at creation. The RX
|
||||||
// it is created. The RX staging sort (overlay/batch) orders packets by
|
// staging sort (overlay/batch) orders packets by (epoch, message counter). A re-handshake never
|
||||||
// (epoch, message counter). A re-handshake never rekeys an existing tunnel; it
|
// rekeys an existing tunnel; it brings up a new hostinfo and ConnectionState with a counter space
|
||||||
// brings up a whole new hostinfo (and ConnectionState) with its own counter
|
// starting near zero, while the old tunnel keeps decrypting until torn down. During that cutover
|
||||||
// space starting near zero, while the old tunnel keeps decrypting in-flight
|
// one flush batch can hold packets from both tunnels, and the epoch keeps the old tunnel's
|
||||||
// packets until it is torn down. During that cutover one flush batch can hold
|
// packets sorted first.
|
||||||
// packets from both tunnels, and the epoch is what keeps the old tunnel's
|
|
||||||
// packets sorted ahead of the new tunnel's.
|
|
||||||
var sessionEpoch atomic.Uint64
|
var sessionEpoch atomic.Uint64
|
||||||
|
|
||||||
type ConnectionState struct {
|
type ConnectionState struct {
|
||||||
|
|||||||
@@ -78,7 +78,7 @@ func newReadyControl(t *testing.T) (*Control, *fakeDevice, *fakeConn) {
|
|||||||
inside: dev,
|
inside: dev,
|
||||||
outside: conn,
|
outside: conn,
|
||||||
writers: []udp.Conn{conn},
|
writers: []udp.Conn{conn},
|
||||||
batchers: make([]batch.RxBatcher, 1),
|
batchers: make([]*batch.MultiCoalescer, 1),
|
||||||
routines: 1,
|
routines: 1,
|
||||||
hostMap: newHostMap(l),
|
hostMap: newHostMap(l),
|
||||||
lightHouse: lh,
|
lightHouse: lh,
|
||||||
@@ -177,7 +177,7 @@ func TestControl_StartMultiqueueFailureReleases(t *testing.T) {
|
|||||||
inside: dev,
|
inside: dev,
|
||||||
outside: conn,
|
outside: conn,
|
||||||
writers: []udp.Conn{conn},
|
writers: []udp.Conn{conn},
|
||||||
batchers: make([]batch.RxBatcher, 2),
|
batchers: make([]*batch.MultiCoalescer, 2),
|
||||||
routines: 2,
|
routines: 2,
|
||||||
l: test.NewLogger(),
|
l: test.NewLogger(),
|
||||||
}
|
}
|
||||||
|
|||||||
+3
-3
@@ -116,11 +116,11 @@ type Interface struct {
|
|||||||
writers []udp.Conn
|
writers []udp.Conn
|
||||||
queues []tio.Queue
|
queues []tio.Queue
|
||||||
// batchers is one per tun queue, wrapping queues[i]. readOutsidePackets
|
// batchers is one per tun queue, wrapping queues[i]. readOutsidePackets
|
||||||
// commits plaintext into the batch.RxBatcher; the plaintext is decrypted
|
// commits plaintext into the batcher; the plaintext is decrypted
|
||||||
// in place inside the UDP receive buffers, so listenOut must call Flush
|
// in place inside the UDP receive buffers, so listenOut must call Flush
|
||||||
// at the end of each UDP recvmmsg batch, before those buffers are
|
// at the end of each UDP recvmmsg batch, before those buffers are
|
||||||
// reused (every udp.Conn ListenOut guarantees that ordering).
|
// reused (every udp.Conn ListenOut guarantees that ordering).
|
||||||
batchers []batch.RxBatcher
|
batchers []*batch.MultiCoalescer
|
||||||
wg sync.WaitGroup
|
wg sync.WaitGroup
|
||||||
|
|
||||||
// fatalErr holds the first unexpected reader error that caused shutdown.
|
// fatalErr holds the first unexpected reader error that caused shutdown.
|
||||||
@@ -218,7 +218,7 @@ func NewInterface(ctx context.Context, c *InterfaceConfig) (*Interface, error) {
|
|||||||
routines: c.routines,
|
routines: c.routines,
|
||||||
version: c.version,
|
version: c.version,
|
||||||
writers: make([]udp.Conn, c.routines),
|
writers: make([]udp.Conn, c.routines),
|
||||||
batchers: make([]batch.RxBatcher, c.routines),
|
batchers: make([]*batch.MultiCoalescer, c.routines),
|
||||||
myVpnNetworks: cs.myVpnNetworks,
|
myVpnNetworks: cs.myVpnNetworks,
|
||||||
myVpnNetworksTable: cs.myVpnNetworksTable,
|
myVpnNetworksTable: cs.myVpnNetworksTable,
|
||||||
myVpnAddrs: cs.myVpnAddrs,
|
myVpnAddrs: cs.myVpnAddrs,
|
||||||
|
|||||||
+7
-8
@@ -316,8 +316,8 @@ var (
|
|||||||
|
|
||||||
// newPacket validates and parses the interesting bits for the firewall out of the ip and sub protocol headers
|
// newPacket validates and parses the interesting bits for the firewall out of the ip and sub protocol headers
|
||||||
func newPacket(data []byte, incoming bool, fp *firewall.ParsedPacket) error {
|
func newPacket(data []byte, incoming bool, fp *firewall.ParsedPacket) error {
|
||||||
// fp is reused across packets; reset the parse byproducts here so a
|
// fp is reused across packets; reset the parse byproducts so an early-error return cannot
|
||||||
// parser's early-error return can't leak the previous packet's offsets.
|
// leak the previous packet's offsets.
|
||||||
fp.IPHdrLen = 0
|
fp.IPHdrLen = 0
|
||||||
fp.FragAny = false
|
fp.FragAny = false
|
||||||
if len(data) < 1 {
|
if len(data) < 1 {
|
||||||
@@ -389,8 +389,8 @@ func parseV6(data []byte, incoming bool, fp *firewall.ParsedPacket) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fp.Protocol = uint8(proto)
|
fp.Protocol = uint8(proto)
|
||||||
// offset is the L4 header start: 40 for a plain packet, past the
|
// offset is the L4 header start: 40 for a plain packet, past the extension chain
|
||||||
// extension chain otherwise. The coalescer only accepts 40.
|
// otherwise. The coalescer only accepts 40.
|
||||||
fp.IPHdrLen = offset
|
fp.IPHdrLen = offset
|
||||||
if incoming {
|
if incoming {
|
||||||
fp.RemotePort = binary.BigEndian.Uint16(data[offset : offset+2])
|
fp.RemotePort = binary.BigEndian.Uint16(data[offset : offset+2])
|
||||||
@@ -409,8 +409,7 @@ func parseV6(data []byte, incoming bool, fp *firewall.ParsedPacket) error {
|
|||||||
return ErrIPv6PacketTooShort
|
return ErrIPv6PacketTooShort
|
||||||
}
|
}
|
||||||
|
|
||||||
// Either way this packet is a fragment shape the coalescer must
|
// A fragment shape the coalescer must not touch either way, first fragment included.
|
||||||
// not touch, first fragment included.
|
|
||||||
fp.FragAny = true
|
fp.FragAny = true
|
||||||
|
|
||||||
// Check if this is the first fragment
|
// Check if this is the first fragment
|
||||||
@@ -471,8 +470,8 @@ func parseV4(data []byte, incoming bool, fp *firewall.ParsedPacket) error {
|
|||||||
// Check if this is the second or further fragment of a fragmented packet.
|
// Check if this is the second or further fragment of a fragmented packet.
|
||||||
flagsfrags := binary.BigEndian.Uint16(data[6:8])
|
flagsfrags := binary.BigEndian.Uint16(data[6:8])
|
||||||
fp.Fragment = (flagsfrags & 0x1FFF) != 0
|
fp.Fragment = (flagsfrags & 0x1FFF) != 0
|
||||||
// Any fragmentation at all (MF or offset): first fragments have readable
|
// Any fragmentation at all (MF or offset): first fragments have readable ports for the
|
||||||
// ports for the firewall but must never be coalesced.
|
// firewall but must never be coalesced.
|
||||||
fp.FragAny = (flagsfrags & 0x3fff) != 0
|
fp.FragAny = (flagsfrags & 0x3fff) != 0
|
||||||
fp.IPHdrLen = ihl
|
fp.IPHdrLen = ihl
|
||||||
|
|
||||||
|
|||||||
+4
-29
@@ -1,36 +1,11 @@
|
|||||||
package batch
|
package batch
|
||||||
|
|
||||||
import "github.com/slackhq/nebula/firewall"
|
|
||||||
|
|
||||||
// SortKey identifies a packet's position in its sender's transmission order.
|
// SortKey identifies a packet's position in its sender's transmission order.
|
||||||
// Epoch is a receiver-local ordinal for the tunnel (ConnectionState) that
|
// Epoch is a receiver-local ordinal for the tunnel (ConnectionState) that decrypted the packet:
|
||||||
// decrypted the packet. A re-handshake replaces the tunnel outright — new
|
// a re-handshake replaces the tunnel outright and the replacement's epoch is higher,
|
||||||
// hostinfo, new keys, a fresh counter space — and the replacement's epoch is
|
// so the old tunnel's packets sort first during the cutover overlap.
|
||||||
// higher, so during the cutover overlap the old tunnel's packets sort first.
|
// Counter is the packet's AEAD message counter within that tunnel.
|
||||||
// Counter is the packet's AEAD message counter within that tunnel. The replay
|
|
||||||
// window has already rejected duplicates by Commit time, so keys are unique
|
|
||||||
// per tunnel and (Epoch, Counter) is a total order with no ties.
|
|
||||||
type SortKey struct {
|
type SortKey struct {
|
||||||
Epoch uint64
|
Epoch uint64
|
||||||
Counter uint64
|
Counter uint64
|
||||||
}
|
}
|
||||||
|
|
||||||
type RxBatcher interface {
|
|
||||||
// Commit stages pkt to be flushed by the batch. key must carry the
|
|
||||||
// packet's session epoch and message counter; pp must be the firewall's
|
|
||||||
// parse of this same packet. The caller must keep pkt valid until the
|
|
||||||
// next Flush, and not re-use it. pp, by contrast, is borrowed only for
|
|
||||||
// the duration of the call — the caller reuses one ParsedPacket per
|
|
||||||
// receive loop — so implementations must copy what they need from it.
|
|
||||||
Commit(pkt []byte, key SortKey, pp *firewall.ParsedPacket) error
|
|
||||||
// Flush emits every staged packet. Packets are first sorted by key, so
|
|
||||||
// within each protocol lane emission follows the sender's transmission
|
|
||||||
// order regardless of arrival order. One shape may legally be overtaken
|
|
||||||
// by later same-flow data: a pure TCP ACK, which does not close its
|
|
||||||
// flow's open coalesce chain (a late ACK is just a stale ACK). Cross-lane
|
|
||||||
// order (TCP vs UDP vs everything else) is not preserved.
|
|
||||||
// Returns the first error observed; keeps draining so one bad packet
|
|
||||||
// doesn't hold up the rest.
|
|
||||||
// After Flush returns, committed payload slices may be recycled.
|
|
||||||
Flush() error
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ type flowKey struct {
|
|||||||
// so this matches a typical carrier-side recvmmsg batch on the UDP socket.
|
// so this matches a typical carrier-side recvmmsg batch on the UDP socket.
|
||||||
const initialSlots = 64
|
const initialSlots = 64
|
||||||
|
|
||||||
// parsedIP is the IP-level result of parseIPPrologue.
|
// parsedIP is the IP-level result of the prologue parsers.
|
||||||
// The caller layers L4-specific parsing (TCP / UDP) on top.
|
// The caller layers L4-specific parsing (TCP / UDP) on top.
|
||||||
type parsedIP struct {
|
type parsedIP struct {
|
||||||
fk flowKey
|
fk flowKey
|
||||||
@@ -31,46 +31,11 @@ type parsedIP struct {
|
|||||||
pkt []byte
|
pkt []byte
|
||||||
}
|
}
|
||||||
|
|
||||||
// parseIPPrologue extracts the IP-level fields the coalescers care about:
|
// parseIPAt validates the IP header for lane parsing. newPacket already resolved the L4 protocol
|
||||||
// IHL/payload length, version, src/dst addresses, and the L4 protocol byte.
|
// and offset for the firewall, so there is no proto sniff here; the caller's ipHdrLen is
|
||||||
// Returns ok=false for malformed input, IPv4 with options or fragmentation,
|
// cross-checked instead. A plain header (v4 IHL 20, v6 exactly 40) is the only coalesceable
|
||||||
// or IPv6 with extension headers (all rejected by both coalescers in
|
// shape. The v6 check is load-bearing: it rejects extension-header packets whose L4 is not at
|
||||||
// identical ways before this refactor).
|
// byte 40. On success p.pkt is trimmed to the IP-declared length.
|
||||||
//
|
|
||||||
// On success, p.pkt is len-trimmed to the IP-declared length so callers
|
|
||||||
// don't have to repeat the trim. wantProto is the IANA protocol number to
|
|
||||||
// require (6 for TCP, 17 for UDP); ok=false for any other value.
|
|
||||||
// This is the standalone-lane-Commit entry; the dispatcher path uses
|
|
||||||
// parseIPAt, where the protocol was already resolved upstream.
|
|
||||||
func parseIPPrologue(pkt []byte, wantProto byte) (parsedIP, bool) {
|
|
||||||
var p parsedIP
|
|
||||||
if len(pkt) < 20 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
switch pkt[0] >> 4 {
|
|
||||||
case 4:
|
|
||||||
if pkt[9] != wantProto {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
return parseIPv4Prologue(pkt)
|
|
||||||
case 6:
|
|
||||||
if len(pkt) < 40 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
if pkt[6] != wantProto {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
return parseIPv6Prologue(pkt)
|
|
||||||
}
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
|
|
||||||
// parseIPAt is the dispatcher-path prologue: newPacket already resolved the
|
|
||||||
// L4 protocol and header offset once for the firewall, so the proto sniff is
|
|
||||||
// replaced by a cross-check of the caller's ipHdrLen. A plain header (v4:
|
|
||||||
// IHL 20, v6: exactly 40 — no options, no extension headers) is the only
|
|
||||||
// coalesceable shape, which is the same rule parseIPPrologue enforces
|
|
||||||
// through its own reads.
|
|
||||||
func parseIPAt(pkt []byte, ipHdrLen int) (parsedIP, bool) {
|
func parseIPAt(pkt []byte, ipHdrLen int) (parsedIP, bool) {
|
||||||
var p parsedIP
|
var p parsedIP
|
||||||
if len(pkt) < 20 {
|
if len(pkt) < 20 {
|
||||||
@@ -91,18 +56,16 @@ func parseIPAt(pkt []byte, ipHdrLen int) (parsedIP, bool) {
|
|||||||
return p, false
|
return p, false
|
||||||
}
|
}
|
||||||
|
|
||||||
// parseIPv4Prologue is the shared IPv4 tail of the two prologue entries.
|
// parseIPv4Prologue is the shared IPv4 tail of the prologue entries; the
|
||||||
// The caller has verified len(pkt) >= 20 and either the protocol
|
// caller has verified len(pkt) >= 20 and the version.
|
||||||
// (parseIPPrologue) or the upstream-resolved header length (parseIPAt).
|
|
||||||
func parseIPv4Prologue(pkt []byte) (parsedIP, bool) {
|
func parseIPv4Prologue(pkt []byte) (parsedIP, bool) {
|
||||||
var p parsedIP
|
var p parsedIP
|
||||||
ihl := int(pkt[0]&0x0f) * 4
|
ihl := int(pkt[0]&0x0f) * 4
|
||||||
if ihl != 20 {
|
if ihl != 20 {
|
||||||
return p, false
|
return p, false
|
||||||
}
|
}
|
||||||
// Reject actual fragmentation (MF or non-zero frag offset). On the
|
// Reject any fragmentation (MF or nonzero offset). The dispatcher already gated FragAny; kept
|
||||||
// dispatcher path FragAny was already gated; kept as defense in depth —
|
// as defense in depth, since a fragment folded into a superpacket would corrupt reassembly.
|
||||||
// a fragment folded into a superpacket would corrupt reassembly.
|
|
||||||
if binary.BigEndian.Uint16(pkt[6:8])&0x3fff != 0 {
|
if binary.BigEndian.Uint16(pkt[6:8])&0x3fff != 0 {
|
||||||
return p, false
|
return p, false
|
||||||
}
|
}
|
||||||
@@ -118,8 +81,8 @@ func parseIPv4Prologue(pkt []byte) (parsedIP, bool) {
|
|||||||
return p, true
|
return p, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// parseIPv6Prologue is the shared IPv6 tail; caller has verified
|
// parseIPv6Prologue is the shared IPv6 tail; the caller has verified
|
||||||
// len(pkt) >= 40 and version/proto-or-offset.
|
// len(pkt) >= 40, the version, and that the L4 header sits at byte 40.
|
||||||
func parseIPv6Prologue(pkt []byte) (parsedIP, bool) {
|
func parseIPv6Prologue(pkt []byte) (parsedIP, bool) {
|
||||||
var p parsedIP
|
var p parsedIP
|
||||||
payloadLen := int(binary.BigEndian.Uint16(pkt[4:6]))
|
payloadLen := int(binary.BigEndian.Uint16(pkt[4:6]))
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
package batch
|
package batch
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"cmp"
|
||||||
"errors"
|
"errors"
|
||||||
"io"
|
"io"
|
||||||
"log/slog"
|
"log/slog"
|
||||||
@@ -9,47 +10,34 @@ import (
|
|||||||
"github.com/slackhq/nebula/firewall"
|
"github.com/slackhq/nebula/firewall"
|
||||||
)
|
)
|
||||||
|
|
||||||
// MultiCoalescer stages plaintext packets with their (epoch, counter) sort
|
// MultiCoalescer stages plaintext packets with their (epoch, counter) sort keys and, at Flush,
|
||||||
// keys, and at Flush replays them in sender-transmission order into
|
// replays them in sender-transmission order into lane-specific batchers selected by L4 protocol.
|
||||||
// lane-specific batchers selected by the IP/L4 protocol of the packet.
|
|
||||||
//
|
//
|
||||||
// Sorting *before* the lanes see anything is what makes the ordering story
|
// Sorting before dispatch keeps the ordering story simple: each lane consumes packets in
|
||||||
// simple: each lane consumes packets in transmission order, builds its slots
|
// transmission order, builds slots in that order, and emits them in creation order. Wire reorder
|
||||||
// in that order, and emits them in creation order. Wire reorder inside a
|
// inside a flush batch is repaired here, before it can fragment a lane's coalesce chains, so the
|
||||||
// flush batch is repaired here, before it can fragment a lane's coalesce
|
// lanes carry no reorder-repair machinery.
|
||||||
// chains, so the lanes carry no reorder-repair machinery of their own.
|
|
||||||
//
|
//
|
||||||
// The ordering contract is per-tunnel transmission order within each lane:
|
// The contract is per-tunnel transmission order within each lane, with two exceptions: a pure TCP
|
||||||
// a sender's packets are emitted in the order it encrypted them. Two
|
// ACK may be overtaken by later same-flow data (it does not close the flow's open chain; a late
|
||||||
// qualifications:
|
// ACK is just a stale ACK), and an unparseable shape seals every open chain in its lane (its flow
|
||||||
// - a pure TCP ACK may be overtaken by later same-flow data, because it
|
// is unknown) and rides the lane as an in-lane verbatim, still in transmission order. Routing
|
||||||
// does not close the flow's open coalesce chain (a late ACK is just a
|
// follows the flow: a flow's non-coalesceable shapes ride its protocol lane rather than falling
|
||||||
// stale ACK; see TCPCoalescer.commitParsed);
|
// to the later-flushed pt lane.
|
||||||
// - an unparseable shape (fragment, IP options) seals every open chain in
|
|
||||||
// its lane — its flow is unknowable, so this is the only way to keep
|
|
||||||
// later data from extending a chain that would emit ahead of it. The
|
|
||||||
// packet then rides its lane as an in-lane passthrough, still in
|
|
||||||
// transmission order.
|
|
||||||
//
|
//
|
||||||
// Routing follows the flow, not the coalesceability: IPv4 fragments keep
|
// Cross-lane order (TCP vs UDP vs everything else) is not preserved.
|
||||||
// their L4 proto visible and IPv6 extension chains are walked to the
|
|
||||||
// terminal proto, so a flow's non-coalesceable shapes ride its lane rather
|
|
||||||
// than falling to the later-flushed pt lane.
|
|
||||||
//
|
|
||||||
// Cross-lane order is intentionally NOT preserved across the TCP/UDP/verbatim split.
|
|
||||||
type MultiCoalescer struct {
|
type MultiCoalescer struct {
|
||||||
tcp *TCPCoalescer
|
tcp *TCPCoalescer
|
||||||
udp *UDPCoalescer
|
udp *UDPCoalescer
|
||||||
pt *Passthrough
|
pt *Passthrough
|
||||||
|
|
||||||
// staged holds this batch's packets and sort keys until Flush. Borrowed:
|
// staged holds this batch's packets and sort keys until Flush. Borrowed: the caller keeps
|
||||||
// the caller keeps each pkt alive until Flush returns.
|
// each pkt alive until Flush returns.
|
||||||
staged []stagedPacket
|
staged []stagedPacket
|
||||||
}
|
}
|
||||||
|
|
||||||
// stagedPacket also carries the scalars dispatch needs from the firewall's
|
// stagedPacket carries the scalars dispatch needs from the firewall's ParsedPacket, copied by
|
||||||
// ParsedPacket: pp itself is reused by the caller per packet and must not be
|
// value: pp is reused by the caller per packet and must not be retained past Commit.
|
||||||
// retained past Commit, so the relevant fields are copied by value here.
|
|
||||||
type stagedPacket struct {
|
type stagedPacket struct {
|
||||||
pkt []byte
|
pkt []byte
|
||||||
key SortKey
|
key SortKey
|
||||||
@@ -58,10 +46,10 @@ type stagedPacket struct {
|
|||||||
ipHdrLen uint16
|
ipHdrLen uint16
|
||||||
}
|
}
|
||||||
|
|
||||||
// NewMultiCoalescer builds a multi-lane batcher over w, based on available
|
// NewMultiCoalescer builds a multi-lane batcher over w, based on available protocol support. The
|
||||||
// protocol support. The staging sort applies even when no GSO lane is
|
// staging sort applies even when no GSO lane is available: passthrough-only platforms still get
|
||||||
// available: passthrough-only platforms still get transmission-order repair.
|
// transmission-order repair.
|
||||||
func NewMultiCoalescer(w io.Writer, l *slog.Logger) RxBatcher {
|
func NewMultiCoalescer(w io.Writer, l *slog.Logger) *MultiCoalescer {
|
||||||
m := &MultiCoalescer{
|
m := &MultiCoalescer{
|
||||||
pt: NewPassthrough(w),
|
pt: NewPassthrough(w),
|
||||||
staged: make([]stagedPacket, 0, initialSlots),
|
staged: make([]stagedPacket, 0, initialSlots),
|
||||||
@@ -71,10 +59,10 @@ func NewMultiCoalescer(w io.Writer, l *slog.Logger) RxBatcher {
|
|||||||
return m
|
return m
|
||||||
}
|
}
|
||||||
|
|
||||||
// Commit stages pkt for the next Flush. All lane dispatch is deferred to
|
// Commit stages pkt for the next Flush; dispatch is deferred so it runs on packets already in
|
||||||
// Flush so it runs on packets already in transmission order. pp is the
|
// transmission order. key carries the packet's tunnel epoch and message counter. pkt is borrowed:
|
||||||
// firewall's parse of pkt — the single source of truth for the packet's
|
// the caller must keep it valid until the next Flush and not re-use it. pp is the firewall's
|
||||||
// protocol and L4 offset — and is only borrowed for this call.
|
// parse of pkt and is borrowed only for this call, so the fields dispatch needs are copied here.
|
||||||
func (m *MultiCoalescer) Commit(pkt []byte, key SortKey, pp *firewall.ParsedPacket) error {
|
func (m *MultiCoalescer) Commit(pkt []byte, key SortKey, pp *firewall.ParsedPacket) error {
|
||||||
m.staged = append(m.staged, stagedPacket{
|
m.staged = append(m.staged, stagedPacket{
|
||||||
pkt: pkt,
|
pkt: pkt,
|
||||||
@@ -86,24 +74,12 @@ func (m *MultiCoalescer) Commit(pkt []byte, key SortKey, pp *firewall.ParsedPack
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// compareStaged orders staged packets by (epoch, counter): sender
|
// compareStaged orders staged packets by (epoch, counter)
|
||||||
// transmission order within a tunnel, tunnel-creation order across a
|
|
||||||
// re-handshake cutover. Keys are unique (see SortKey), so this is a total
|
|
||||||
// order and sort stability doesn't matter.
|
|
||||||
func compareStaged(a, b stagedPacket) int {
|
func compareStaged(a, b stagedPacket) int {
|
||||||
if a.key.Epoch != b.key.Epoch {
|
if c := cmp.Compare(a.key.Epoch, b.key.Epoch); c != 0 {
|
||||||
if a.key.Epoch < b.key.Epoch {
|
return c
|
||||||
return -1
|
|
||||||
}
|
|
||||||
return 1
|
|
||||||
}
|
}
|
||||||
if a.key.Counter == b.key.Counter {
|
return cmp.Compare(a.key.Counter, b.key.Counter)
|
||||||
return 0
|
|
||||||
}
|
|
||||||
if a.key.Counter < b.key.Counter {
|
|
||||||
return -1
|
|
||||||
}
|
|
||||||
return 1
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// dispatch routes one staged packet to its lane.
|
// dispatch routes one staged packet to its lane.
|
||||||
@@ -145,11 +121,12 @@ func (m *MultiCoalescer) dispatch(sp stagedPacket) error {
|
|||||||
return m.pt.enqueue(sp.pkt)
|
return m.pt.enqueue(sp.pkt)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Flush sorts the staged batch into transmission order, replays it into the
|
// Flush sorts the staged batch into transmission order, replays it into the lanes, then flushes each lane.
|
||||||
// lanes, then flushes each lane.
|
// Drains everything and returns the joined errors; one bad packet does not hold up the rest.
|
||||||
|
// After Flush returns, committed payload slices may be recycled.
|
||||||
func (m *MultiCoalescer) Flush() error {
|
func (m *MultiCoalescer) Flush() error {
|
||||||
// Arrival order is already almost sorted (reorder is the exception, not
|
// Arrival order is already almost sorted (reorder is the exception), which pdqsort detects
|
||||||
// the rule), which pdqsort detects and handles in near-linear time.
|
// and handles in near-linear time.
|
||||||
slices.SortFunc(m.staged, compareStaged)
|
slices.SortFunc(m.staged, compareStaged)
|
||||||
|
|
||||||
var errs []error
|
var errs []error
|
||||||
|
|||||||
@@ -21,16 +21,10 @@ func (k *keySeq) next() SortKey {
|
|||||||
return SortKey{Epoch: k.epoch, Counter: k.counter}
|
return SortKey{Epoch: k.epoch, Counter: k.counter}
|
||||||
}
|
}
|
||||||
|
|
||||||
// newTestMultiCoalescer builds a batcher over w and asserts the concrete
|
// newTestMultiCoalescer builds a batcher over w.
|
||||||
// type so tests can reach into the lanes.
|
|
||||||
func newTestMultiCoalescer(tb testing.TB, w io.Writer) *MultiCoalescer {
|
func newTestMultiCoalescer(tb testing.TB, w io.Writer) *MultiCoalescer {
|
||||||
tb.Helper()
|
tb.Helper()
|
||||||
b := NewMultiCoalescer(w, test.NewLogger())
|
return NewMultiCoalescer(w, test.NewLogger())
|
||||||
m, ok := b.(*MultiCoalescer)
|
|
||||||
if !ok {
|
|
||||||
tb.Fatalf("want a *MultiCoalescer, got %T", b)
|
|
||||||
}
|
|
||||||
return m
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestMultiCoalescerRoutesByProto confirms TCP/UDP/other land in the right
|
// TestMultiCoalescerRoutesByProto confirms TCP/UDP/other land in the right
|
||||||
|
|||||||
@@ -2,11 +2,10 @@ package batch
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"io"
|
"io"
|
||||||
|
|
||||||
"github.com/slackhq/nebula/firewall"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// Passthrough is a RxBatcher that doesn't batch anything, it just accumulates and then sends packets.
|
// Passthrough is MultiCoalescer's verbatim lane: no batching, packets are written at Flush in the
|
||||||
|
// order enqueued.
|
||||||
type Passthrough struct {
|
type Passthrough struct {
|
||||||
out io.Writer
|
out io.Writer
|
||||||
slots [][]byte
|
slots [][]byte
|
||||||
@@ -19,12 +18,7 @@ func NewPassthrough(w io.Writer) *Passthrough {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (p *Passthrough) Commit(pkt []byte, _ SortKey, _ *firewall.ParsedPacket) error {
|
// enqueue accepts one packet, already sorted into transmission order by dispatch.
|
||||||
return p.enqueue(pkt)
|
|
||||||
}
|
|
||||||
|
|
||||||
// enqueue is the lane-facing half of Commit: MultiCoalescer.dispatch hands
|
|
||||||
// packets here already sorted into transmission order.
|
|
||||||
func (p *Passthrough) enqueue(pkt []byte) error {
|
func (p *Passthrough) enqueue(pkt []byte) error {
|
||||||
p.slots = append(p.slots, pkt)
|
p.slots = append(p.slots, pkt)
|
||||||
return nil
|
return nil
|
||||||
|
|||||||
+70
-147
@@ -26,24 +26,21 @@ const tcpCoalesceMaxSegs = 64
|
|||||||
// into. IPv6 (40) + TCP with full options (60) = 100 bytes.
|
// into. IPv6 (40) + TCP with full options (60) = 100 bytes.
|
||||||
const tcpCoalesceHdrCap = 100
|
const tcpCoalesceHdrCap = 100
|
||||||
|
|
||||||
// coalesceSlot is one entry in the coalescer's ordered event queue.
|
// coalesceSlot is one entry in the coalescer's ordered event queue. A verbatim slot holds a single
|
||||||
// When verbatim is true the slot holds a single borrowed packet that is
|
// borrowed packet emitted as-is (pure ACK, non-admissible TCP, unparseable, or oversize seed); a
|
||||||
// emitted as-is (pure ACK, non-admissible TCP, unparseable, or oversize seed).
|
// non-verbatim slot is an in-progress coalesced superpacket. payIovs are borrowed slices of the
|
||||||
// When verbatim is false the slot is an in-progress coalesced superpacket.
|
// caller's plaintext buffers; the caller must keep them alive until Flush.
|
||||||
// hdrBuf is a mutable copy of the seed's IP+TCP header, populated on the
|
|
||||||
// first append (we patch total length and pseudo-header partial at flush;
|
|
||||||
// a slot that never grows flushes from rawPkt and never touches hdrBuf)
|
|
||||||
// payIovs are *borrowed* slices from the caller's plaintext buffers.
|
|
||||||
// The caller (listenOut) must keep those buffers alive until Flush.
|
|
||||||
type coalesceSlot struct {
|
type coalesceSlot struct {
|
||||||
verbatim bool
|
verbatim bool
|
||||||
// rawPkt is borrowed: the whole packet for verbatim slots, the seed
|
// rawPkt is borrowed: the whole packet for verbatim slots, the seed packet for coalesce
|
||||||
// packet for coalesce slots. A coalesce slot that never grows past one
|
// slots. A slot that never grows past one segment is emitted from rawPkt so its original
|
||||||
// segment is emitted from rawPkt so its original (already valid) L4
|
// (already valid) L4 checksum ships DATA_VALID instead of making the kernel recompute it.
|
||||||
// checksum ships DATA_VALID instead of making the kernel recompute it.
|
|
||||||
rawPkt []byte
|
rawPkt []byte
|
||||||
|
|
||||||
fk flowKey
|
fk flowKey
|
||||||
|
// hdrBuf is a mutable copy of the seed's IP+TCP header, populated on the first append. Total
|
||||||
|
// length and the pseudo-header checksum partial are patched at flush. A slot that never grows
|
||||||
|
// flushes from rawPkt and never touches hdrBuf.
|
||||||
hdrBuf [tcpCoalesceHdrCap]byte
|
hdrBuf [tcpCoalesceHdrCap]byte
|
||||||
hdrLen int
|
hdrLen int
|
||||||
ipHdrLen int
|
ipHdrLen int
|
||||||
@@ -55,24 +52,20 @@ type coalesceSlot struct {
|
|||||||
payIovs [][]byte
|
payIovs [][]byte
|
||||||
}
|
}
|
||||||
|
|
||||||
// TCPCoalescer accumulates adjacent in-flow TCP data segments across multiple concurrent flows
|
// TCPCoalescer accumulates adjacent in-flow TCP data segments across multiple concurrent flows and
|
||||||
// and emits each flow's run as a single TSO superpacket via tio.GSOWriter.
|
// emits each flow's run as a single TSO superpacket via tio.GSOWriter. Input must be in sender
|
||||||
// It expects its input in sender-transmission order (MultiCoalescer sorts the
|
// transmission order (MultiCoalescer sorts by (epoch, counter) before dispatch); slots are emitted
|
||||||
// staged batch by (epoch, counter) before dispatching here) and emits slots in
|
// in creation order, so emission reproduces transmission order except for the pure-ACK case in
|
||||||
// creation order, which therefore reproduces transmission order — modulo the
|
// commitParsed. Owns no locks; one coalescer per TUN write queue.
|
||||||
// pure-ACK allowance in commitParsed.
|
|
||||||
// Owns no locks; one coalescer per TUN write queue.
|
|
||||||
type TCPCoalescer struct {
|
type TCPCoalescer struct {
|
||||||
w tio.GSOWriter
|
w tio.GSOWriter
|
||||||
|
|
||||||
// slots is the ordered event queue. Flush walks it once and emits each
|
// slots is the ordered event queue. Flush walks it once and emits each
|
||||||
// entry as either a WriteGSO (coalesced) or a w.Write (verbatim).
|
// entry as either a WriteGSO (coalesced) or a w.Write (verbatim).
|
||||||
slots []*coalesceSlot
|
slots []*coalesceSlot
|
||||||
// openSlots maps a flow key to its still-open slot, so new segments can
|
// openSlots maps a flow key to its open slot so new segments can extend an in-progress
|
||||||
// extend an in-progress superpacket in O(1). Membership here is what
|
// superpacket in O(1). Removal is what closes a chain: on PSH or a short last segment, on a
|
||||||
// keeps a chain extendable: slots are removed when they close (PSH or
|
// non-admissible packet for the flow, or in Flush.
|
||||||
// short-last-segment), when a non-admissible packet for that flow
|
|
||||||
// arrives, or in Flush.
|
|
||||||
openSlots map[flowKey]*coalesceSlot
|
openSlots map[flowKey]*coalesceSlot
|
||||||
// lastSlot caches the most recently touched open slot. Bulk traffic
|
// lastSlot caches the most recently touched open slot. Bulk traffic
|
||||||
// arrives in same-flow runs (single-flow steady state, or GRO bursts
|
// arrives in same-flow runs (single-flow steady state, or GRO bursts
|
||||||
@@ -104,28 +97,17 @@ func NewTCPCoalescer(w io.Writer, l *slog.Logger) *TCPCoalescer {
|
|||||||
// parsedTCP holds the fields extracted from a single parse so later steps
|
// parsedTCP holds the fields extracted from a single parse so later steps
|
||||||
// (admission, slot lookup, canAppend) don't re-walk the header.
|
// (admission, slot lookup, canAppend) don't re-walk the header.
|
||||||
type parsedTCP struct {
|
type parsedTCP struct {
|
||||||
fk flowKey
|
fk flowKey
|
||||||
ipHdrLen int
|
ipHdrLen int
|
||||||
tcpHdrLen int
|
hdrLen int
|
||||||
hdrLen int
|
payLen int
|
||||||
payLen int
|
seq uint32
|
||||||
seq uint32
|
flags byte
|
||||||
flags byte
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// parseTCPBase extracts the flow key and IP/TCP offsets for any TCP packet,
|
// parseTCPAt extracts the flow key and IP/TCP offsets for a packet the dispatcher already knows is
|
||||||
// regardless of whether it's admissible for coalescing. Returns ok=false for non-TCP or malformed input.
|
// TCP; ipHdrLen is the upstream-resolved L4 offset (see parseIPAt). Returns ok=false for malformed
|
||||||
// Accepts IPv4 (no options or fragmentation) and IPv6 (no extension headers).
|
// input or any shape that must not coalesce (IPv4 options/fragmentation, IPv6 extension headers).
|
||||||
func parseTCPBase(pkt []byte) (parsedTCP, bool) {
|
|
||||||
ip, ok := parseIPPrologue(pkt, ipProtoTCP)
|
|
||||||
if !ok {
|
|
||||||
return parsedTCP{}, false
|
|
||||||
}
|
|
||||||
return parseTCPTail(ip)
|
|
||||||
}
|
|
||||||
|
|
||||||
// parseTCPAt is parseTCPBase for the dispatcher path: the packet is already
|
|
||||||
// known to be TCP and ipHdrLen is the upstream-resolved L4 offset (see parseIPAt).
|
|
||||||
func parseTCPAt(pkt []byte, ipHdrLen int) (parsedTCP, bool) {
|
func parseTCPAt(pkt []byte, ipHdrLen int) (parsedTCP, bool) {
|
||||||
ip, ok := parseIPAt(pkt, ipHdrLen)
|
ip, ok := parseIPAt(pkt, ipHdrLen)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -151,7 +133,6 @@ func parseTCPTail(ip parsedIP) (parsedTCP, bool) {
|
|||||||
if len(pkt) < p.ipHdrLen+tcpOff {
|
if len(pkt) < p.ipHdrLen+tcpOff {
|
||||||
return p, false
|
return p, false
|
||||||
}
|
}
|
||||||
p.tcpHdrLen = tcpOff
|
|
||||||
p.hdrLen = p.ipHdrLen + tcpOff
|
p.hdrLen = p.ipHdrLen + tcpOff
|
||||||
p.payLen = len(pkt) - p.hdrLen
|
p.payLen = len(pkt) - p.hdrLen
|
||||||
p.fk.sport = binary.BigEndian.Uint16(pkt[p.ipHdrLen : p.ipHdrLen+2])
|
p.fk.sport = binary.BigEndian.Uint16(pkt[p.ipHdrLen : p.ipHdrLen+2])
|
||||||
@@ -161,85 +142,31 @@ func parseTCPTail(ip parsedIP) (parsedTCP, bool) {
|
|||||||
return p, true
|
return p, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// TCP flag bits (byte 13 of the TCP header). Only the bits actually consulted
|
// TCP flag bits (byte 13 of the TCP header). Only the bits the coalescer consults are named;
|
||||||
// by the coalescer are named; FIN/SYN/RST/URG/CWR are rejected via the
|
// FIN/SYN/RST/URG/CWR are rejected by the negative mask in commitParsed.
|
||||||
// negative mask in coalesceable, not by name.
|
|
||||||
const (
|
const (
|
||||||
tcpFlagPsh = 0x08
|
tcpFlagPsh = 0x08
|
||||||
tcpFlagAck = 0x10
|
tcpFlagAck = 0x10
|
||||||
tcpFlagEce = 0x40
|
tcpFlagEce = 0x40
|
||||||
)
|
)
|
||||||
|
|
||||||
// coalesceable reports whether a parsed TCP segment is eligible for
|
// sealAllOpen closes every open coalesce chain. Called for unparseable packets: the flow key is
|
||||||
// coalescing. Accepts ACK, ACK|PSH, ACK|ECE, ACK|PSH|ECE with a
|
// unknown, so any open chain could otherwise absorb later data and emit it ahead of this packet.
|
||||||
// non-empty payload. CWR is excluded because it marks a one-shot
|
|
||||||
// congestion-window-reduced transition the receiver must observe at a
|
|
||||||
// segment boundary.
|
|
||||||
func (p parsedTCP) coalesceable() bool {
|
|
||||||
if p.flags&tcpFlagAck == 0 {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if p.flags&^(tcpFlagAck|tcpFlagPsh|tcpFlagEce) != 0 {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
return p.payLen > 0
|
|
||||||
}
|
|
||||||
|
|
||||||
// pureAck reports whether a parsed segment is a bare acknowledgment: no
|
|
||||||
// payload and nothing beyond ACK|PSH|ECE in the flags. These are the only
|
|
||||||
// non-coalesceable shape that may safely pass through WITHOUT sealing the
|
|
||||||
// flow's open slot — a late-delivered stale ACK is ignored by the receiver,
|
|
||||||
// whereas SYN/FIN/RST/CWR mark transitions the flow must observe in order.
|
|
||||||
func (p parsedTCP) pureAck() bool {
|
|
||||||
return p.payLen == 0 &&
|
|
||||||
p.flags&tcpFlagAck != 0 &&
|
|
||||||
p.flags&^(tcpFlagAck|tcpFlagPsh|tcpFlagEce) == 0
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *TCPCoalescer) Commit(pkt []byte) error {
|
|
||||||
info, ok := parseTCPBase(pkt)
|
|
||||||
if !ok {
|
|
||||||
// Unparseable shape: flow key unknowable, so seal every open chain to
|
|
||||||
// keep later data from extending a chain that would emit ahead of it.
|
|
||||||
c.sealAllOpen()
|
|
||||||
c.addVerbatim(pkt)
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
return c.commitParsed(pkt, info)
|
|
||||||
}
|
|
||||||
|
|
||||||
// sealAllOpen closes every open coalesce chain: nothing committed after this
|
|
||||||
// call can extend a slot created before it. Called when an unparseable packet
|
|
||||||
// arrives — its flow is unknown, so any open chain might be the one whose
|
|
||||||
// later data would otherwise leapfrog it.
|
|
||||||
func (c *TCPCoalescer) sealAllOpen() {
|
func (c *TCPCoalescer) sealAllOpen() {
|
||||||
clear(c.openSlots)
|
clear(c.openSlots)
|
||||||
c.lastSlot = nil
|
c.lastSlot = nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// commitParsed is the post-parse half of Commit. The caller must have
|
// commitParsed commits one parsed TCP packet. The caller (dispatch, via parseTCPAt) supplies a
|
||||||
// already verified parseTCPBase succeeded (info is a valid TCP parse).
|
// valid parse so the header is not re-walked here.
|
||||||
// Used by MultiCoalescer.Commit to avoid re-walking the IP/TCP header
|
|
||||||
// after the dispatcher has already done so.
|
|
||||||
func (c *TCPCoalescer) commitParsed(pkt []byte, info parsedTCP) error {
|
func (c *TCPCoalescer) commitParsed(pkt []byte, info parsedTCP) error {
|
||||||
if !info.coalesceable() {
|
// Admission: only ACK, ACK|PSH, ACK|ECE, ACK|PSH|ECE may ride a coalesce chain. CWR marks a
|
||||||
if info.pureAck() {
|
// one-shot congestion transition the receiver must observe at a segment boundary. NB: AccECN
|
||||||
// A bare window/ack update carries no ordering obligation toward
|
// reuses CWR as ACE counter bits; revisit this check if inner hosts adopt AccECN.
|
||||||
// the flow's data: delivering it after later-transmitted data only
|
if info.flags&tcpFlagAck == 0 || info.flags&^(tcpFlagAck|tcpFlagPsh|tcpFlagEce) != 0 {
|
||||||
// makes it a stale ACK, which receivers ignore. Skipping the
|
// SYN/FIN/RST/URG/CWR must be observed in sequence. Seal the flow's open slot so later
|
||||||
// evict keeps a bidirectional flow's inbound data run coalescing
|
// in-flow packets cannot extend it and emit ahead of this verbatim. The len guard skips
|
||||||
// across the peer ACKs interleaved into it — kernel GRO likewise
|
// hashing the 38-byte key on ack-dominant queues, where the map is almost always empty.
|
||||||
// doesn't flush held data on a pure ACK. This is the one place
|
|
||||||
// emission can deviate from transmission order.
|
|
||||||
c.addVerbatim(pkt)
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
// TCP but not admissible (SYN/FIN/RST/URG/CWR or a shape the flow
|
|
||||||
// must observe in sequence). Seal this flow's open slot so later
|
|
||||||
// in-flow packets don't extend it and emit ahead of this verbatim;
|
|
||||||
// with input in transmission order that pins the verbatim's exact
|
|
||||||
// in-flow position. The len guard skips hashing the 38-byte key on
|
|
||||||
// ack-dominant queues, where the map is almost always empty.
|
|
||||||
if len(c.openSlots) != 0 {
|
if len(c.openSlots) != 0 {
|
||||||
if last := c.lastSlot; last != nil && last.fk == info.fk {
|
if last := c.lastSlot; last != nil && last.fk == info.fk {
|
||||||
c.lastSlot = nil
|
c.lastSlot = nil
|
||||||
@@ -249,6 +176,14 @@ func (c *TCPCoalescer) commitParsed(pkt []byte, info parsedTCP) error {
|
|||||||
c.addVerbatim(pkt)
|
c.addVerbatim(pkt)
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
if info.payLen == 0 {
|
||||||
|
// Pure ACK: no ordering obligation toward the flow's data. Delivering it after
|
||||||
|
// later-transmitted data only makes it a stale ACK, which receivers ignore. Not sealing
|
||||||
|
// keeps a bidirectional flow's data run coalescing across interleaved peer ACKs, matching
|
||||||
|
// kernel GRO. This is the only place emission deviates from transmission order.
|
||||||
|
c.addVerbatim(pkt)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// Cached-slot fast path. Arrival isn't per-packet interleaved even with
|
// Cached-slot fast path. Arrival isn't per-packet interleaved even with
|
||||||
// many flows: wire-side GRO delivers runs of same-flow packets
|
// many flows: wire-side GRO delivers runs of same-flow packets
|
||||||
@@ -291,11 +226,9 @@ func (c *TCPCoalescer) Flush() error {
|
|||||||
for _, s := range c.slots {
|
for _, s := range c.slots {
|
||||||
var err error
|
var err error
|
||||||
if s.verbatim || s.numSeg == 1 {
|
if s.verbatim || s.numSeg == 1 {
|
||||||
// A slot that never grew (nor absorbed a merge) is byte-identical
|
// A slot that never grew is byte-identical to its seed packet; ship the original so
|
||||||
// to the packet it was seeded from; ship the original so its valid
|
// its valid checksum rides the DATA_VALID path instead of a kernel software csum.
|
||||||
// checksum rides the DATA_VALID path instead of paying a kernel
|
// appendPayload only touches hdrBuf once numSeg >= 2, so rawPkt is pristine here.
|
||||||
// software csum. appendPayload only touches hdrBuf once
|
|
||||||
// numSeg >= 2, so rawPkt is still pristine here.
|
|
||||||
_, err = c.w.Write(s.rawPkt)
|
_, err = c.w.Write(s.rawPkt)
|
||||||
} else {
|
} else {
|
||||||
err = c.flushSlot(s)
|
err = c.flushSlot(s)
|
||||||
@@ -328,8 +261,8 @@ func (c *TCPCoalescer) seed(pkt []byte, info parsedTCP) {
|
|||||||
}
|
}
|
||||||
s := c.take()
|
s := c.take()
|
||||||
s.verbatim = false
|
s.verbatim = false
|
||||||
// rawPkt serves the numSeg==1 fast path in Flush and is the header
|
// rawPkt serves the numSeg==1 fast path in Flush and is the header source for canAppend until
|
||||||
// source for canAppend until the first append copies it into hdrBuf.
|
// the first append copies it into hdrBuf.
|
||||||
s.rawPkt = pkt
|
s.rawPkt = pkt
|
||||||
s.hdrLen = info.hdrLen
|
s.hdrLen = info.hdrLen
|
||||||
s.ipHdrLen = info.ipHdrLen
|
s.ipHdrLen = info.ipHdrLen
|
||||||
@@ -345,21 +278,16 @@ func (c *TCPCoalescer) seed(pkt []byte, info parsedTCP) {
|
|||||||
c.openSlots[info.fk] = s
|
c.openSlots[info.fk] = s
|
||||||
c.lastSlot = s
|
c.lastSlot = s
|
||||||
} else if last := c.lastSlot; last != nil && last.fk == info.fk {
|
} else if last := c.lastSlot; last != nil && last.fk == info.fk {
|
||||||
// PSH-on-seed closes the chain immediately: never registered as
|
// PSH on the seed closes the chain immediately; it is never registered as open. Drop any
|
||||||
// open. Any prior cached open slot for this flow has just been
|
// stale cache entry for this flow too.
|
||||||
// closed-and-replaced by this seed, so drop the cache too.
|
|
||||||
c.lastSlot = nil
|
c.lastSlot = nil
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// canAppend reports whether info's packet extends the slot's seed: same
|
// canAppend reports whether info's packet extends the slot's seed: same header shape and stable
|
||||||
// header shape and stable contents, adjacent seq, not oversized. A closed
|
// contents, adjacent seq, not oversized. A closed chain never reaches here; closing removes the
|
||||||
// chain never reaches here — closing removes the slot from openSlots, and
|
// slot from openSlots, the only path in. Header reads use rawPkt because hdrBuf is populated
|
||||||
// openSlots/lastSlot are the only paths in.
|
// lazily on the first append; every field consulted here is one the pre-flush patches never touch.
|
||||||
// Header reads go through rawPkt, not hdrBuf: hdrBuf is populated lazily on
|
|
||||||
// the first append, and every field consulted here is one the pre-flush
|
|
||||||
// patches never touch (headersMatch skips the flags byte, and PSH is the
|
|
||||||
// only bit patched before flush).
|
|
||||||
func (c *TCPCoalescer) canAppend(s *coalesceSlot, pkt []byte, info parsedTCP) bool {
|
func (c *TCPCoalescer) canAppend(s *coalesceSlot, pkt []byte, info parsedTCP) bool {
|
||||||
if info.hdrLen != s.hdrLen {
|
if info.hdrLen != s.hdrLen {
|
||||||
return false
|
return false
|
||||||
@@ -391,14 +319,13 @@ func (c *TCPCoalescer) canAppend(s *coalesceSlot, pkt []byte, info parsedTCP) bo
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// appendPayload folds info's packet into s and reports whether the chain is
|
// appendPayload folds info's packet into s and reports whether the chain is now closed: the
|
||||||
// now closed: the segment was sub-gsoSize (kernel TSO allows only the final
|
// segment was sub-gsoSize (kernel TSO allows only the final segment to be short) or carried PSH.
|
||||||
// segment to be short) or carried PSH (a semantic delimiter). The caller
|
// The caller must deregister a closed slot from openSlots.
|
||||||
// must deregister a closed slot from openSlots.
|
|
||||||
func (c *TCPCoalescer) appendPayload(s *coalesceSlot, pkt []byte, info parsedTCP) bool {
|
func (c *TCPCoalescer) appendPayload(s *coalesceSlot, pkt []byte, info parsedTCP) bool {
|
||||||
if s.numSeg == 1 {
|
if s.numSeg == 1 {
|
||||||
// First append: populate hdrBuf from the seed packet. Deferred out
|
// First append: populate hdrBuf from the seed. Deferred out of seed so solo slots, which
|
||||||
// of seed so solo slots, which flush from rawPkt, never pay the copy.
|
// flush from rawPkt, never pay the copy.
|
||||||
copy(s.hdrBuf[:s.hdrLen], s.rawPkt[:s.hdrLen])
|
copy(s.hdrBuf[:s.hdrLen], s.rawPkt[:s.hdrLen])
|
||||||
}
|
}
|
||||||
s.payIovs = append(s.payIovs, pkt[info.hdrLen:info.hdrLen+info.payLen])
|
s.payIovs = append(s.payIovs, pkt[info.hdrLen:info.hdrLen+info.payLen])
|
||||||
@@ -406,8 +333,7 @@ func (c *TCPCoalescer) appendPayload(s *coalesceSlot, pkt []byte, info parsedTCP
|
|||||||
s.totalPay += info.payLen
|
s.totalPay += info.payLen
|
||||||
s.nextSeq = info.seq + uint32(info.payLen)
|
s.nextSeq = info.seq + uint32(info.payLen)
|
||||||
if info.flags&tcpFlagPsh != 0 {
|
if info.flags&tcpFlagPsh != 0 {
|
||||||
// Propagate PSH into the seed header so kernel TSO sets it on the
|
// Propagate PSH into the seed header so kernel TSO sets it on the last segment.
|
||||||
// last segment. Without this the sender's push signal is dropped.
|
|
||||||
s.hdrBuf[s.ipHdrLen+13] |= tcpFlagPsh
|
s.hdrBuf[s.ipHdrLen+13] |= tcpFlagPsh
|
||||||
}
|
}
|
||||||
return info.payLen < s.gsoSize || info.flags&tcpFlagPsh != 0
|
return info.payLen < s.gsoSize || info.flags&tcpFlagPsh != 0
|
||||||
@@ -497,13 +423,10 @@ func headersMatch(a, b []byte, isV6 bool, ipHdrLen int) bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// logSeqGaps reports same-flow seq discontinuities between consecutively
|
// logSeqGaps reports same-flow seq discontinuities between consecutively created data slots. Input
|
||||||
// created data slots. Input arrives in transmission order (MultiCoalescer
|
// is in transmission order, so a gap is traffic this batch never contained: loss upstream of
|
||||||
// sorts by (epoch, counter) before dispatch), so a gap here is traffic this
|
// nebula, reorder across a flush boundary, or a retransmit (negative gap). The caller gates on
|
||||||
// batch never contained: loss upstream of nebula, a reorder spanning a flush
|
// debug level, so the map only allocates when enabled.
|
||||||
// boundary (which no intra-batch mechanism can repair), or a retransmit
|
|
||||||
// (negative gap). Logged so the operator can quantify how often that happens.
|
|
||||||
// The caller gates on debug level, so the map only allocates when asked for.
|
|
||||||
func (c *TCPCoalescer) logSeqGaps() {
|
func (c *TCPCoalescer) logSeqGaps() {
|
||||||
prevByFlow := make(map[flowKey]*coalesceSlot, len(c.slots))
|
prevByFlow := make(map[flowKey]*coalesceSlot, len(c.slots))
|
||||||
for _, s := range c.slots {
|
for _, s := range c.slots {
|
||||||
|
|||||||
@@ -83,19 +83,9 @@ type parsedUDP struct {
|
|||||||
payLen int
|
payLen int
|
||||||
}
|
}
|
||||||
|
|
||||||
// parseUDP extracts the flow key and IP/UDP offsets for a UDP packet.
|
// parseUDPAt extracts the flow key and IP/UDP offsets for a packet the dispatcher already knows is
|
||||||
// Returns ok=false for non-UDP, malformed, or unsupported header shapes
|
// UDP; ipHdrLen is the upstream-resolved L4 offset (see parseIPAt). Returns ok=false for malformed
|
||||||
// (IPv4 with options/fragmentation, IPv6 with extension headers).
|
// input or any shape that must not coalesce (IPv4 options/fragmentation, IPv6 extension headers).
|
||||||
func parseUDP(pkt []byte) (parsedUDP, bool) {
|
|
||||||
ip, ok := parseIPPrologue(pkt, ipProtoUDP)
|
|
||||||
if !ok {
|
|
||||||
return parsedUDP{}, false
|
|
||||||
}
|
|
||||||
return parseUDPTail(ip)
|
|
||||||
}
|
|
||||||
|
|
||||||
// parseUDPAt is parseUDP for the dispatcher path: the packet is already
|
|
||||||
// known to be UDP and ipHdrLen is the upstream-resolved L4 offset (see parseIPAt).
|
|
||||||
func parseUDPAt(pkt []byte, ipHdrLen int) (parsedUDP, bool) {
|
func parseUDPAt(pkt []byte, ipHdrLen int) (parsedUDP, bool) {
|
||||||
ip, ok := parseIPAt(pkt, ipHdrLen)
|
ip, ok := parseIPAt(pkt, ipHdrLen)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -126,23 +116,11 @@ func parseUDPTail(ip parsedIP) (parsedUDP, bool) {
|
|||||||
return p, true
|
return p, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// Commit borrows pkt. The caller must keep pkt valid until the next Flush.
|
// commitParsed commits one parsed UDP packet. The caller (dispatch, via parseUDPAt) supplies a
|
||||||
func (c *UDPCoalescer) Commit(pkt []byte) error {
|
// valid parse so the header is not re-walked here.
|
||||||
info, ok := parseUDP(pkt)
|
|
||||||
if !ok {
|
|
||||||
c.addVerbatim(pkt)
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
return c.commitParsed(pkt, info)
|
|
||||||
}
|
|
||||||
|
|
||||||
// commitParsed is the post-parse half of Commit. The caller must have
|
|
||||||
// already verified parseUDP succeeded. Used by MultiCoalescer.Commit to
|
|
||||||
// avoid re-walking the IP/UDP header.
|
|
||||||
func (c *UDPCoalescer) commitParsed(pkt []byte, info parsedUDP) error {
|
func (c *UDPCoalescer) commitParsed(pkt []byte, info parsedUDP) error {
|
||||||
// A zero-length UDP datagram (UDP `length` == 8) is legal and must still
|
// A zero-length UDP datagram (length == 8) is legal and must reach the TUN, but cannot be
|
||||||
// reach the TUN, but it can't be coalesced. The len guard skips hashing
|
// coalesced. The len guard skips hashing the key when no flow is open.
|
||||||
// the key when no flow is open.
|
|
||||||
if info.payLen == 0 {
|
if info.payLen == 0 {
|
||||||
if len(c.openSlots) != 0 {
|
if len(c.openSlots) != 0 {
|
||||||
if last := c.lastSlot; last != nil && last.fk == info.fk {
|
if last := c.lastSlot; last != nil && last.fk == info.fk {
|
||||||
@@ -206,10 +184,8 @@ func (c *UDPCoalescer) Flush() error {
|
|||||||
return first
|
return first
|
||||||
}
|
}
|
||||||
|
|
||||||
// sealAllOpen closes every open coalesce chain: nothing committed after this
|
// sealAllOpen closes every open coalesce chain. Called for unparseable packets: the flow key is
|
||||||
// call can extend a slot created before it. Called when an unparseable packet
|
// unknown, so any open chain could otherwise absorb later data and emit it ahead of this packet.
|
||||||
// arrives — its flow is unknown, so any open chain might be the one whose
|
|
||||||
// later data would otherwise leapfrog it.
|
|
||||||
func (c *UDPCoalescer) sealAllOpen() {
|
func (c *UDPCoalescer) sealAllOpen() {
|
||||||
clear(c.openSlots)
|
clear(c.openSlots)
|
||||||
c.lastSlot = nil
|
c.lastSlot = nil
|
||||||
@@ -229,8 +205,8 @@ func (c *UDPCoalescer) seed(pkt []byte, info parsedUDP) {
|
|||||||
}
|
}
|
||||||
s := c.take()
|
s := c.take()
|
||||||
s.verbatim = false
|
s.verbatim = false
|
||||||
// rawPkt serves the numSeg==1 fast path in Flush and is the header
|
// rawPkt serves the numSeg==1 fast path in Flush and is the header source for canAppend until
|
||||||
// source for canAppend until the first append copies it into hdrBuf.
|
// the first append copies it into hdrBuf.
|
||||||
s.rawPkt = pkt
|
s.rawPkt = pkt
|
||||||
s.hdrLen = info.hdrLen
|
s.hdrLen = info.hdrLen
|
||||||
s.ipHdrLen = info.ipHdrLen
|
s.ipHdrLen = info.ipHdrLen
|
||||||
@@ -261,10 +237,9 @@ func (c *UDPCoalescer) canAppend(s *udpSlot, pkt []byte, info parsedUDP) bool {
|
|||||||
if s.hdrLen+s.totalPay+info.payLen > udpCoalesceBufSize {
|
if s.hdrLen+s.totalPay+info.payLen > udpCoalesceBufSize {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
// Header reads go through rawPkt: hdrBuf is populated lazily on the
|
// Header reads use rawPkt because hdrBuf is populated lazily on the first append; the fields
|
||||||
// first append, and the fields consulted here are never patched before
|
// consulted here are never patched before flush. A closed chain never reaches here; closing
|
||||||
// flush. A closed chain never reaches here — closing removes the slot
|
// removes the slot from openSlots, the only path in.
|
||||||
// from openSlots, the only path in.
|
|
||||||
if !s.isV6 && !ipv4CanCoalesceID(s.rawPkt, pkt, s.numSeg) {
|
if !s.isV6 && !ipv4CanCoalesceID(s.rawPkt, pkt, s.numSeg) {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
@@ -274,14 +249,13 @@ func (c *UDPCoalescer) canAppend(s *udpSlot, pkt []byte, info parsedUDP) bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// appendPayload folds info's packet into s and reports whether the chain is
|
// appendPayload folds info's packet into s and reports whether the chain is now closed: kernel
|
||||||
// now closed: kernel UDP-GSO requires every segment but the last to be
|
// UDP-GSO requires every segment but the last to be exactly gsoSize, so a short segment must be
|
||||||
// exactly gsoSize, so a short segment must be the final one. The caller
|
// the final one. The caller must deregister a closed slot from openSlots.
|
||||||
// must deregister a closed slot from openSlots.
|
|
||||||
func (c *UDPCoalescer) appendPayload(s *udpSlot, pkt []byte, info parsedUDP) bool {
|
func (c *UDPCoalescer) appendPayload(s *udpSlot, pkt []byte, info parsedUDP) bool {
|
||||||
if s.numSeg == 1 {
|
if s.numSeg == 1 {
|
||||||
// First append: populate hdrBuf from the seed packet. Deferred out
|
// First append: populate hdrBuf from the seed. Deferred out of seed so solo slots, which
|
||||||
// of seed so solo slots, which flush from rawPkt, never pay the copy.
|
// flush from rawPkt, never pay the copy.
|
||||||
copy(s.hdrBuf[:s.hdrLen], s.rawPkt[:s.hdrLen])
|
copy(s.hdrBuf[:s.hdrLen], s.rawPkt[:s.hdrLen])
|
||||||
}
|
}
|
||||||
s.payIovs = append(s.payIovs, pkt[info.hdrLen:info.hdrLen+info.payLen])
|
s.payIovs = append(s.payIovs, pkt[info.hdrLen:info.hdrLen+info.payLen])
|
||||||
|
|||||||
Reference in New Issue
Block a user