mirror of
https://github.com/slackhq/nebula.git
synced 2026-08-16 01:16:58 +02:00
Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e9357ff426 | |||
| 8c50fc3f60 | |||
| 2f4532f102 |
@@ -1,70 +0,0 @@
|
|||||||
package nebula
|
|
||||||
|
|
||||||
import "net/netip"
|
|
||||||
|
|
||||||
// sendBatchCap is the maximum number of encrypted packets accumulated before a
|
|
||||||
// flush is forced. TSO superpackets segment to at most ~45 packets on
|
|
||||||
// reasonable MTUs, so 128 leaves headroom without bloating the backing
|
|
||||||
// allocation.
|
|
||||||
const sendBatchCap = 128
|
|
||||||
|
|
||||||
// sendBatch accumulates encrypted UDP packets for a single sendmmsg flush.
|
|
||||||
// One sendBatch is owned by each listenIn goroutine; no locking is needed.
|
|
||||||
// The backing storage holds up to batchCap packets of slotCap bytes each;
|
|
||||||
// bufs and dsts are parallel slices of committed slots.
|
|
||||||
type sendBatch struct {
|
|
||||||
bufs [][]byte
|
|
||||||
dsts []netip.AddrPort
|
|
||||||
backing []byte
|
|
||||||
slotCap int
|
|
||||||
batchCap int
|
|
||||||
nextSlot int
|
|
||||||
}
|
|
||||||
|
|
||||||
func newSendBatch(batchCap, slotCap int) *sendBatch {
|
|
||||||
return &sendBatch{
|
|
||||||
bufs: make([][]byte, 0, batchCap),
|
|
||||||
dsts: make([]netip.AddrPort, 0, batchCap),
|
|
||||||
backing: make([]byte, batchCap*slotCap),
|
|
||||||
slotCap: slotCap,
|
|
||||||
batchCap: batchCap,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Next returns a zero-length slice with slotCap capacity over the next unused
|
|
||||||
// slot's backing bytes. The caller writes into the returned slice and then
|
|
||||||
// calls Commit with the final length and destination. Next returns nil when
|
|
||||||
// the batch is full.
|
|
||||||
func (b *sendBatch) Next() []byte {
|
|
||||||
if b.nextSlot >= b.batchCap {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
start := b.nextSlot * b.slotCap
|
|
||||||
return b.backing[start : start : start+b.slotCap]
|
|
||||||
}
|
|
||||||
|
|
||||||
// Commit records the slot just returned by Next as a packet of length n
|
|
||||||
// destined for dst.
|
|
||||||
func (b *sendBatch) Commit(n int, dst netip.AddrPort) {
|
|
||||||
start := b.nextSlot * b.slotCap
|
|
||||||
b.bufs = append(b.bufs, b.backing[start:start+n])
|
|
||||||
b.dsts = append(b.dsts, dst)
|
|
||||||
b.nextSlot++
|
|
||||||
}
|
|
||||||
|
|
||||||
// Reset clears committed slots; backing storage is retained for reuse.
|
|
||||||
func (b *sendBatch) Reset() {
|
|
||||||
b.bufs = b.bufs[:0]
|
|
||||||
b.dsts = b.dsts[:0]
|
|
||||||
b.nextSlot = 0
|
|
||||||
}
|
|
||||||
|
|
||||||
// Len returns the number of committed packets.
|
|
||||||
func (b *sendBatch) Len() int {
|
|
||||||
return len(b.bufs)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Cap returns the maximum number of slots in the batch.
|
|
||||||
func (b *sendBatch) Cap() int {
|
|
||||||
return b.batchCap
|
|
||||||
}
|
|
||||||
-137
@@ -1,137 +0,0 @@
|
|||||||
package nebula
|
|
||||||
|
|
||||||
import (
|
|
||||||
"net/netip"
|
|
||||||
"testing"
|
|
||||||
)
|
|
||||||
|
|
||||||
func TestSendBatchBookkeeping(t *testing.T) {
|
|
||||||
b := newSendBatch(4, 32)
|
|
||||||
if b.Len() != 0 || b.Cap() != 4 {
|
|
||||||
t.Fatalf("fresh batch: len=%d cap=%d", b.Len(), b.Cap())
|
|
||||||
}
|
|
||||||
|
|
||||||
ap := netip.MustParseAddrPort("10.0.0.1:4242")
|
|
||||||
for i := 0; i < 4; i++ {
|
|
||||||
slot := b.Next()
|
|
||||||
if slot == nil {
|
|
||||||
t.Fatalf("slot %d: Next returned nil before cap", i)
|
|
||||||
}
|
|
||||||
if cap(slot) != 32 || len(slot) != 0 {
|
|
||||||
t.Fatalf("slot %d: got len=%d cap=%d want len=0 cap=32", i, len(slot), cap(slot))
|
|
||||||
}
|
|
||||||
// Write a marker byte.
|
|
||||||
slot = append(slot, byte(i), byte(i+1), byte(i+2))
|
|
||||||
b.Commit(len(slot), ap)
|
|
||||||
}
|
|
||||||
if b.Next() != nil {
|
|
||||||
t.Fatalf("Next should return nil when full")
|
|
||||||
}
|
|
||||||
if b.Len() != 4 {
|
|
||||||
t.Fatalf("Len=%d want 4", b.Len())
|
|
||||||
}
|
|
||||||
for i, buf := range b.bufs {
|
|
||||||
if len(buf) != 3 || buf[0] != byte(i) {
|
|
||||||
t.Errorf("buf %d: %x", i, buf)
|
|
||||||
}
|
|
||||||
if b.dsts[i] != ap {
|
|
||||||
t.Errorf("dst %d: got %v want %v", i, b.dsts[i], ap)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Reset returns empty and Next works again.
|
|
||||||
b.Reset()
|
|
||||||
if b.Len() != 0 {
|
|
||||||
t.Fatalf("after Reset Len=%d want 0", b.Len())
|
|
||||||
}
|
|
||||||
slot := b.Next()
|
|
||||||
if slot == nil || cap(slot) != 32 {
|
|
||||||
t.Fatalf("after Reset Next nil or wrong cap: %v cap=%d", slot == nil, cap(slot))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestBatchSegmentable(t *testing.T) {
|
|
||||||
ap := netip.MustParseAddrPort("10.0.0.1:4242")
|
|
||||||
other := netip.MustParseAddrPort("10.0.0.2:4242")
|
|
||||||
|
|
||||||
mk := func(addrs []netip.AddrPort, sizes []int) *sendBatch {
|
|
||||||
b := newSendBatch(len(addrs), 64)
|
|
||||||
for i, a := range addrs {
|
|
||||||
s := b.Next()
|
|
||||||
for j := 0; j < sizes[i]; j++ {
|
|
||||||
s = append(s, byte(j))
|
|
||||||
}
|
|
||||||
b.Commit(len(s), a)
|
|
||||||
}
|
|
||||||
return b
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Run("uniform same dst", func(t *testing.T) {
|
|
||||||
b := mk([]netip.AddrPort{ap, ap, ap}, []int{10, 10, 10})
|
|
||||||
seg, ok := batchSegmentable(b)
|
|
||||||
if !ok || seg != 10 {
|
|
||||||
t.Fatalf("got seg=%d ok=%v", seg, ok)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("last segment short ok", func(t *testing.T) {
|
|
||||||
b := mk([]netip.AddrPort{ap, ap, ap}, []int{10, 10, 4})
|
|
||||||
seg, ok := batchSegmentable(b)
|
|
||||||
if !ok || seg != 10 {
|
|
||||||
t.Fatalf("got seg=%d ok=%v", seg, ok)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("mixed dst rejected", func(t *testing.T) {
|
|
||||||
b := mk([]netip.AddrPort{ap, other, ap}, []int{10, 10, 10})
|
|
||||||
if _, ok := batchSegmentable(b); ok {
|
|
||||||
t.Fatalf("expected rejection for mixed dst")
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("mid-batch short rejected", func(t *testing.T) {
|
|
||||||
b := mk([]netip.AddrPort{ap, ap, ap}, []int{10, 4, 10})
|
|
||||||
if _, ok := batchSegmentable(b); ok {
|
|
||||||
t.Fatalf("expected rejection for short mid-batch")
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("mid-batch longer rejected", func(t *testing.T) {
|
|
||||||
b := mk([]netip.AddrPort{ap, ap, ap}, []int{10, 11, 10})
|
|
||||||
if _, ok := batchSegmentable(b); ok {
|
|
||||||
t.Fatalf("expected rejection for longer mid-batch")
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("last longer rejected", func(t *testing.T) {
|
|
||||||
b := mk([]netip.AddrPort{ap, ap, ap}, []int{10, 10, 11})
|
|
||||||
if _, ok := batchSegmentable(b); ok {
|
|
||||||
t.Fatalf("expected rejection for longer last segment")
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("first zero rejected", func(t *testing.T) {
|
|
||||||
b := mk([]netip.AddrPort{ap, ap}, []int{0, 10})
|
|
||||||
if _, ok := batchSegmentable(b); ok {
|
|
||||||
t.Fatalf("expected rejection for zero first")
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSendBatchSlotsDoNotOverlap(t *testing.T) {
|
|
||||||
b := newSendBatch(3, 8)
|
|
||||||
ap := netip.MustParseAddrPort("10.0.0.1:80")
|
|
||||||
|
|
||||||
// Fill three slots, each with its own sentinel byte.
|
|
||||||
for i := 0; i < 3; i++ {
|
|
||||||
s := b.Next()
|
|
||||||
s = append(s, byte(0xA0+i), byte(0xB0+i))
|
|
||||||
b.Commit(len(s), ap)
|
|
||||||
}
|
|
||||||
|
|
||||||
for i, buf := range b.bufs {
|
|
||||||
if buf[0] != byte(0xA0+i) || buf[1] != byte(0xB0+i) {
|
|
||||||
t.Errorf("slot %d corrupted: %x", i, buf)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -50,9 +50,15 @@ func main() {
|
|||||||
os.Exit(0)
|
os.Exit(0)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
l := logrus.New()
|
||||||
|
l.Out = os.Stdout
|
||||||
|
|
||||||
if *serviceFlag != "" {
|
if *serviceFlag != "" {
|
||||||
doService(configPath, configTest, Build, serviceFlag)
|
if err := doService(configPath, configTest, Build, serviceFlag); err != nil {
|
||||||
os.Exit(1)
|
l.WithError(err).Error("Service command failed")
|
||||||
|
os.Exit(1)
|
||||||
|
}
|
||||||
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
if *configPath == "" {
|
if *configPath == "" {
|
||||||
@@ -61,9 +67,6 @@ func main() {
|
|||||||
os.Exit(1)
|
os.Exit(1)
|
||||||
}
|
}
|
||||||
|
|
||||||
l := logrus.New()
|
|
||||||
l.Out = os.Stdout
|
|
||||||
|
|
||||||
c := config.NewC(l)
|
c := config.NewC(l)
|
||||||
err := c.Load(*configPath)
|
err := c.Load(*configPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -57,11 +57,11 @@ func fileExists(filename string) bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func doService(configPath *string, configTest *bool, build string, serviceFlag *string) {
|
func doService(configPath *string, configTest *bool, build string, serviceFlag *string) error {
|
||||||
if *configPath == "" {
|
if *configPath == "" {
|
||||||
ex, err := os.Executable()
|
ex, err := os.Executable()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
panic(err)
|
return err
|
||||||
}
|
}
|
||||||
*configPath = filepath.Dir(ex) + "/config.yaml"
|
*configPath = filepath.Dir(ex) + "/config.yaml"
|
||||||
if !fileExists(*configPath) {
|
if !fileExists(*configPath) {
|
||||||
@@ -88,13 +88,13 @@ func doService(configPath *string, configTest *bool, build string, serviceFlag *
|
|||||||
// - above, in `Run` we create a `logrus.Logger` which is what nebula expects to use
|
// - above, in `Run` we create a `logrus.Logger` which is what nebula expects to use
|
||||||
s, err := service.New(prg, svcConfig)
|
s, err := service.New(prg, svcConfig)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
log.Fatal(err)
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
errs := make(chan error, 5)
|
errs := make(chan error, 5)
|
||||||
logger, err = s.Logger(errs)
|
logger, err = s.Logger(errs)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
log.Fatal(err)
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
go func() {
|
go func() {
|
||||||
@@ -109,18 +109,16 @@ func doService(configPath *string, configTest *bool, build string, serviceFlag *
|
|||||||
|
|
||||||
switch *serviceFlag {
|
switch *serviceFlag {
|
||||||
case "run":
|
case "run":
|
||||||
err = s.Run()
|
if err := s.Run(); err != nil {
|
||||||
if err != nil {
|
|
||||||
// Route any errors to the system logger
|
// Route any errors to the system logger
|
||||||
logger.Error(err)
|
logger.Error(err)
|
||||||
}
|
}
|
||||||
default:
|
default:
|
||||||
err := service.Control(s, *serviceFlag)
|
if err := service.Control(s, *serviceFlag); err != nil {
|
||||||
if err != nil {
|
|
||||||
log.Printf("Valid actions: %q\n", service.ControlAction)
|
log.Printf("Valid actions: %q\n", service.ControlAction)
|
||||||
log.Fatal(err)
|
return err
|
||||||
}
|
}
|
||||||
return
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return nil
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -10,7 +10,6 @@ import (
|
|||||||
"github.com/flynn/noise"
|
"github.com/flynn/noise"
|
||||||
"github.com/slackhq/nebula/cert"
|
"github.com/slackhq/nebula/cert"
|
||||||
"github.com/slackhq/nebula/config"
|
"github.com/slackhq/nebula/config"
|
||||||
"github.com/slackhq/nebula/overlay"
|
|
||||||
"github.com/slackhq/nebula/test"
|
"github.com/slackhq/nebula/test"
|
||||||
"github.com/slackhq/nebula/udp"
|
"github.com/slackhq/nebula/udp"
|
||||||
"github.com/stretchr/testify/assert"
|
"github.com/stretchr/testify/assert"
|
||||||
@@ -53,7 +52,7 @@ func Test_NewConnectionManagerTest(t *testing.T) {
|
|||||||
lh := newTestLighthouse()
|
lh := newTestLighthouse()
|
||||||
ifce := &Interface{
|
ifce := &Interface{
|
||||||
hostMap: hostMap,
|
hostMap: hostMap,
|
||||||
inside: &overlay.NoopTun{},
|
inside: &test.NoopTun{},
|
||||||
outside: &udp.NoopConn{},
|
outside: &udp.NoopConn{},
|
||||||
firewall: &Firewall{},
|
firewall: &Firewall{},
|
||||||
lightHouse: lh,
|
lightHouse: lh,
|
||||||
@@ -136,7 +135,7 @@ func Test_NewConnectionManagerTest2(t *testing.T) {
|
|||||||
lh := newTestLighthouse()
|
lh := newTestLighthouse()
|
||||||
ifce := &Interface{
|
ifce := &Interface{
|
||||||
hostMap: hostMap,
|
hostMap: hostMap,
|
||||||
inside: &overlay.NoopTun{},
|
inside: &test.NoopTun{},
|
||||||
outside: &udp.NoopConn{},
|
outside: &udp.NoopConn{},
|
||||||
firewall: &Firewall{},
|
firewall: &Firewall{},
|
||||||
lightHouse: lh,
|
lightHouse: lh,
|
||||||
@@ -221,7 +220,7 @@ func Test_NewConnectionManager_DisconnectInactive(t *testing.T) {
|
|||||||
lh := newTestLighthouse()
|
lh := newTestLighthouse()
|
||||||
ifce := &Interface{
|
ifce := &Interface{
|
||||||
hostMap: hostMap,
|
hostMap: hostMap,
|
||||||
inside: &overlay.NoopTun{},
|
inside: &test.NoopTun{},
|
||||||
outside: &udp.NoopConn{},
|
outside: &udp.NoopConn{},
|
||||||
firewall: &Firewall{},
|
firewall: &Firewall{},
|
||||||
lightHouse: lh,
|
lightHouse: lh,
|
||||||
@@ -348,7 +347,7 @@ func Test_NewConnectionManagerTest_DisconnectInvalid(t *testing.T) {
|
|||||||
lh := newTestLighthouse()
|
lh := newTestLighthouse()
|
||||||
ifce := &Interface{
|
ifce := &Interface{
|
||||||
hostMap: hostMap,
|
hostMap: hostMap,
|
||||||
inside: &overlay.NoopTun{},
|
inside: &test.NoopTun{},
|
||||||
outside: &udp.NoopConn{},
|
outside: &udp.NoopConn{},
|
||||||
firewall: &Firewall{},
|
firewall: &Firewall{},
|
||||||
lightHouse: lh,
|
lightHouse: lh,
|
||||||
|
|||||||
+31
@@ -11,6 +11,7 @@ import (
|
|||||||
|
|
||||||
"github.com/sirupsen/logrus"
|
"github.com/sirupsen/logrus"
|
||||||
"github.com/slackhq/nebula/cert"
|
"github.com/slackhq/nebula/cert"
|
||||||
|
"github.com/slackhq/nebula/firewall/events"
|
||||||
"github.com/slackhq/nebula/header"
|
"github.com/slackhq/nebula/header"
|
||||||
"github.com/slackhq/nebula/overlay"
|
"github.com/slackhq/nebula/overlay"
|
||||||
)
|
)
|
||||||
@@ -340,6 +341,36 @@ func (c *Control) Device() overlay.Device {
|
|||||||
return c.f.inside
|
return c.f.inside
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// SetFirewallEventReporter installs an event reporter on the current firewall.
|
||||||
|
// Passing nil clears any installed reporter. The reporter is carried across
|
||||||
|
// firewall rule reloads. Report* methods are invoked while nebula holds
|
||||||
|
// internal locks and must be non-blocking; in particular they must not call
|
||||||
|
// back into *Control methods that touch the firewall, or deadlock will
|
||||||
|
// result.
|
||||||
|
//
|
||||||
|
// Installation is performed by shallow-copying the current *Firewall,
|
||||||
|
// setting the reporter field on the copy, and swapping the pointer under
|
||||||
|
// the conntrack lock. Every Firewall the data path sees therefore has an
|
||||||
|
// immutable reporter slot, and emit sites can read it without any
|
||||||
|
// synchronization of their own.
|
||||||
|
func (c *Control) SetFirewallEventReporter(r events.Reporter) {
|
||||||
|
old := c.f.firewall
|
||||||
|
if old == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
old.Conntrack.Lock()
|
||||||
|
defer old.Conntrack.Unlock()
|
||||||
|
|
||||||
|
// Re-read under the lock in case a concurrent reload swapped in a new
|
||||||
|
// Firewall between the unlocked load above and here. Both Firewalls share
|
||||||
|
// the same Conntrack pointer in the normal (non-overflow) reload path,
|
||||||
|
// so the lock we hold is the right one for whichever we see now.
|
||||||
|
current := c.f.firewall
|
||||||
|
fw := *current
|
||||||
|
fw.reporter = r
|
||||||
|
c.f.firewall = &fw
|
||||||
|
}
|
||||||
|
|
||||||
func copyHostInfo(h *HostInfo, preferredRanges []netip.Prefix) ControlHostInfo {
|
func copyHostInfo(h *HostInfo, preferredRanges []netip.Prefix) ControlHostInfo {
|
||||||
chi := ControlHostInfo{
|
chi := ControlHostInfo{
|
||||||
VpnAddrs: make([]netip.Addr, len(h.vpnAddrs)),
|
VpnAddrs: make([]netip.Addr, len(h.vpnAddrs)),
|
||||||
|
|||||||
+203
-52
@@ -1,12 +1,14 @@
|
|||||||
package nebula
|
package nebula
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
"net"
|
"net"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
|
"sync/atomic"
|
||||||
|
|
||||||
"github.com/gaissmai/bart"
|
"github.com/gaissmai/bart"
|
||||||
"github.com/miekg/dns"
|
"github.com/miekg/dns"
|
||||||
@@ -14,32 +16,207 @@ import (
|
|||||||
"github.com/slackhq/nebula/config"
|
"github.com/slackhq/nebula/config"
|
||||||
)
|
)
|
||||||
|
|
||||||
// This whole thing should be rewritten to use context
|
type dnsServer struct {
|
||||||
|
|
||||||
var dnsR *dnsRecords
|
|
||||||
var dnsServer *dns.Server
|
|
||||||
var dnsAddr string
|
|
||||||
|
|
||||||
type dnsRecords struct {
|
|
||||||
sync.RWMutex
|
sync.RWMutex
|
||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
|
ctx context.Context
|
||||||
dnsMap4 map[string]netip.Addr
|
dnsMap4 map[string]netip.Addr
|
||||||
dnsMap6 map[string]netip.Addr
|
dnsMap6 map[string]netip.Addr
|
||||||
hostMap *HostMap
|
hostMap *HostMap
|
||||||
myVpnAddrsTable *bart.Lite
|
myVpnAddrsTable *bart.Lite
|
||||||
|
|
||||||
|
mux *dns.ServeMux
|
||||||
|
|
||||||
|
// enabled mirrors `lighthouse.serve_dns && lighthouse.am_lighthouse`.
|
||||||
|
// Start, Add, and reload consult it so callers don't need to know the
|
||||||
|
// gating rules. When it toggles off via reload, accumulated records are
|
||||||
|
// cleared so a later re-enable starts with a fresh map populated from
|
||||||
|
// new handshakes.
|
||||||
|
enabled atomic.Bool
|
||||||
|
|
||||||
|
serverMu sync.Mutex
|
||||||
|
server *dns.Server
|
||||||
|
// started is closed once `server` has finished binding (or after
|
||||||
|
// ListenAndServe returns on a bind failure). Stop waits on it before
|
||||||
|
// calling Shutdown to avoid the miekg/dns "server not started" race
|
||||||
|
// where a Shutdown that arrives before bind completes is silently
|
||||||
|
// ignored, leaving the listener running forever.
|
||||||
|
started chan struct{}
|
||||||
|
addr string
|
||||||
}
|
}
|
||||||
|
|
||||||
func newDnsRecords(l *logrus.Logger, cs *CertState, hostMap *HostMap) *dnsRecords {
|
// newDnsServerFromConfig builds a dnsServer, applies the initial config, and
|
||||||
return &dnsRecords{
|
// registers a reload callback. The reload callback is registered before the
|
||||||
|
// initial config is applied, so a SIGHUP can later enable, fix, or disable
|
||||||
|
// DNS even if the initial application failed.
|
||||||
|
//
|
||||||
|
// The dnsServer internally gates on `lighthouse.serve_dns &&
|
||||||
|
// lighthouse.am_lighthouse`. Start and Add are safe to call unconditionally,
|
||||||
|
// they no-op when DNS isn't enabled. Each Start invocation owns a ctx-cancel
|
||||||
|
// watcher that tears the listener down on nebula shutdown. The returned
|
||||||
|
// pointer is always non-nil, even on error.
|
||||||
|
func newDnsServerFromConfig(ctx context.Context, l *logrus.Logger, cs *CertState, hostMap *HostMap, c *config.C) (*dnsServer, error) {
|
||||||
|
ds := &dnsServer{
|
||||||
l: l,
|
l: l,
|
||||||
|
ctx: ctx,
|
||||||
dnsMap4: make(map[string]netip.Addr),
|
dnsMap4: make(map[string]netip.Addr),
|
||||||
dnsMap6: make(map[string]netip.Addr),
|
dnsMap6: make(map[string]netip.Addr),
|
||||||
hostMap: hostMap,
|
hostMap: hostMap,
|
||||||
myVpnAddrsTable: cs.myVpnAddrsTable,
|
myVpnAddrsTable: cs.myVpnAddrsTable,
|
||||||
}
|
}
|
||||||
|
ds.mux = dns.NewServeMux()
|
||||||
|
ds.mux.HandleFunc(".", ds.handleDnsRequest)
|
||||||
|
|
||||||
|
c.RegisterReloadCallback(func(c *config.C) {
|
||||||
|
if err := ds.reload(c, false); err != nil {
|
||||||
|
l.WithError(err).Error("Failed to reload DNS responder from config")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
if err := ds.reload(c, true); err != nil {
|
||||||
|
return ds, err
|
||||||
|
}
|
||||||
|
return ds, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (d *dnsRecords) Query(q uint16, data string) netip.Addr {
|
// reload applies the latest config and reconciles the running state with it:
|
||||||
|
// - enabled toggled on -> spawn a runner
|
||||||
|
// - enabled toggled off -> stop the runner
|
||||||
|
// - listen address changed (while running) -> restart on the new address
|
||||||
|
// - everything else -> no-op
|
||||||
|
//
|
||||||
|
// On the initial call it only records configuration; Control.Start is what
|
||||||
|
// launches the first runner via dnsStart.
|
||||||
|
func (d *dnsServer) reload(c *config.C, initial bool) error {
|
||||||
|
wantsDns := c.GetBool("lighthouse.serve_dns", false)
|
||||||
|
amLighthouse := c.GetBool("lighthouse.am_lighthouse", false)
|
||||||
|
enabled := wantsDns && amLighthouse
|
||||||
|
newAddr := getDnsServerAddr(c)
|
||||||
|
|
||||||
|
d.serverMu.Lock()
|
||||||
|
running := d.server
|
||||||
|
runningStarted := d.started
|
||||||
|
sameAddr := d.addr == newAddr
|
||||||
|
d.addr = newAddr
|
||||||
|
d.enabled.Store(enabled)
|
||||||
|
d.serverMu.Unlock()
|
||||||
|
|
||||||
|
if initial {
|
||||||
|
if wantsDns && !amLighthouse {
|
||||||
|
d.l.Warn("DNS server refusing to run because this host is not a lighthouse.")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if !enabled {
|
||||||
|
if running != nil {
|
||||||
|
d.Stop()
|
||||||
|
}
|
||||||
|
// Drop any records that accumulated while enabled; a later re-enable
|
||||||
|
// will repopulate from fresh handshakes.
|
||||||
|
d.clearRecords()
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if running == nil {
|
||||||
|
// Was disabled (or never started); bring it up now.
|
||||||
|
go d.Start()
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if sameAddr {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
d.shutdownServer(running, runningStarted, "reload")
|
||||||
|
// Old Start goroutine has now exited; bring up a fresh listener on the
|
||||||
|
// new address.
|
||||||
|
go d.Start()
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// shutdownServer waits for the server to finish binding (so Shutdown actually
|
||||||
|
// stops it rather than no-oping) and then shuts it down.
|
||||||
|
func (d *dnsServer) shutdownServer(srv *dns.Server, started chan struct{}, reason string) {
|
||||||
|
if srv == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if started != nil {
|
||||||
|
<-started
|
||||||
|
}
|
||||||
|
if err := srv.Shutdown(); err != nil {
|
||||||
|
d.l.WithError(err).WithField("reason", reason).Warn("Failed to shut down the DNS responder")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Start binds and serves the DNS responder. Blocks until Stop is called or
|
||||||
|
// the listener errors. Safe to call when DNS is disabled (returns
|
||||||
|
// immediately). This is what Control.dnsStart points at.
|
||||||
|
//
|
||||||
|
// Must be invoked after the tun device is active so that lighthouse.dns.host
|
||||||
|
// may bind to a nebula IP.
|
||||||
|
func (d *dnsServer) Start() {
|
||||||
|
if !d.enabled.Load() {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
started := make(chan struct{})
|
||||||
|
d.serverMu.Lock()
|
||||||
|
if d.ctx.Err() != nil {
|
||||||
|
d.serverMu.Unlock()
|
||||||
|
return
|
||||||
|
}
|
||||||
|
addr := d.addr
|
||||||
|
server := &dns.Server{
|
||||||
|
Addr: addr,
|
||||||
|
Net: "udp",
|
||||||
|
Handler: d.mux,
|
||||||
|
NotifyStartedFunc: func() { close(started) },
|
||||||
|
}
|
||||||
|
d.server = server
|
||||||
|
d.started = started
|
||||||
|
d.serverMu.Unlock()
|
||||||
|
|
||||||
|
// Per-invocation ctx watcher. Exits when Start does, so we don't leak a
|
||||||
|
// watcher per reload-driven restart.
|
||||||
|
done := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
select {
|
||||||
|
case <-d.ctx.Done():
|
||||||
|
d.shutdownServer(server, started, "shutdown")
|
||||||
|
case <-done:
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
d.l.WithField("dnsListener", addr).Info("Starting DNS responder")
|
||||||
|
err := server.ListenAndServe()
|
||||||
|
close(done)
|
||||||
|
|
||||||
|
// If the listener never bound (bind error) NotifyStartedFunc never fires,
|
||||||
|
// so close started here to release any Stop caller waiting on it.
|
||||||
|
select {
|
||||||
|
case <-started:
|
||||||
|
default:
|
||||||
|
close(started)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
d.l.WithError(err).Warn("Failed to run the DNS responder")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Stop shuts down the active server, if any. Idempotent.
|
||||||
|
func (d *dnsServer) Stop() {
|
||||||
|
d.serverMu.Lock()
|
||||||
|
srv := d.server
|
||||||
|
started := d.started
|
||||||
|
d.server = nil
|
||||||
|
d.started = nil
|
||||||
|
d.serverMu.Unlock()
|
||||||
|
d.shutdownServer(srv, started, "stop")
|
||||||
|
}
|
||||||
|
|
||||||
|
func (d *dnsServer) Query(q uint16, data string) netip.Addr {
|
||||||
data = strings.ToLower(data)
|
data = strings.ToLower(data)
|
||||||
d.RLock()
|
d.RLock()
|
||||||
defer d.RUnlock()
|
defer d.RUnlock()
|
||||||
@@ -57,7 +234,7 @@ func (d *dnsRecords) Query(q uint16, data string) netip.Addr {
|
|||||||
return netip.Addr{}
|
return netip.Addr{}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (d *dnsRecords) QueryCert(data string) string {
|
func (d *dnsServer) QueryCert(data string) string {
|
||||||
ip, err := netip.ParseAddr(data[:len(data)-1])
|
ip, err := netip.ParseAddr(data[:len(data)-1])
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return ""
|
return ""
|
||||||
@@ -80,8 +257,19 @@ func (d *dnsRecords) QueryCert(data string) string {
|
|||||||
return string(b)
|
return string(b)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// clearRecords drops all DNS records.
|
||||||
|
func (d *dnsServer) clearRecords() {
|
||||||
|
d.Lock()
|
||||||
|
defer d.Unlock()
|
||||||
|
clear(d.dnsMap4)
|
||||||
|
clear(d.dnsMap6)
|
||||||
|
}
|
||||||
|
|
||||||
// Add adds the first IPv4 and IPv6 address that appears in `addresses` as the record for `host`
|
// Add adds the first IPv4 and IPv6 address that appears in `addresses` as the record for `host`
|
||||||
func (d *dnsRecords) Add(host string, addresses []netip.Addr) {
|
func (d *dnsServer) Add(host string, addresses []netip.Addr) {
|
||||||
|
if !d.enabled.Load() {
|
||||||
|
return
|
||||||
|
}
|
||||||
host = strings.ToLower(host)
|
host = strings.ToLower(host)
|
||||||
d.Lock()
|
d.Lock()
|
||||||
defer d.Unlock()
|
defer d.Unlock()
|
||||||
@@ -101,7 +289,7 @@ func (d *dnsRecords) Add(host string, addresses []netip.Addr) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (d *dnsRecords) isSelfNebulaOrLocalhost(addr string) bool {
|
func (d *dnsServer) isSelfNebulaOrLocalhost(addr string) bool {
|
||||||
a, _, _ := net.SplitHostPort(addr)
|
a, _, _ := net.SplitHostPort(addr)
|
||||||
b, err := netip.ParseAddr(a)
|
b, err := netip.ParseAddr(a)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -116,7 +304,7 @@ func (d *dnsRecords) isSelfNebulaOrLocalhost(addr string) bool {
|
|||||||
return d.myVpnAddrsTable.Contains(b)
|
return d.myVpnAddrsTable.Contains(b)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (d *dnsRecords) parseQuery(m *dns.Msg, w dns.ResponseWriter) {
|
func (d *dnsServer) parseQuery(m *dns.Msg, w dns.ResponseWriter) {
|
||||||
for _, q := range m.Question {
|
for _, q := range m.Question {
|
||||||
switch q.Qtype {
|
switch q.Qtype {
|
||||||
case dns.TypeA, dns.TypeAAAA:
|
case dns.TypeA, dns.TypeAAAA:
|
||||||
@@ -150,7 +338,7 @@ func (d *dnsRecords) parseQuery(m *dns.Msg, w dns.ResponseWriter) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (d *dnsRecords) handleDnsRequest(w dns.ResponseWriter, r *dns.Msg) {
|
func (d *dnsServer) handleDnsRequest(w dns.ResponseWriter, r *dns.Msg) {
|
||||||
m := new(dns.Msg)
|
m := new(dns.Msg)
|
||||||
m.SetReply(r)
|
m.SetReply(r)
|
||||||
m.Compress = false
|
m.Compress = false
|
||||||
@@ -163,21 +351,6 @@ func (d *dnsRecords) handleDnsRequest(w dns.ResponseWriter, r *dns.Msg) {
|
|||||||
w.WriteMsg(m)
|
w.WriteMsg(m)
|
||||||
}
|
}
|
||||||
|
|
||||||
func dnsMain(l *logrus.Logger, cs *CertState, hostMap *HostMap, c *config.C) func() {
|
|
||||||
dnsR = newDnsRecords(l, cs, hostMap)
|
|
||||||
|
|
||||||
// attach request handler func
|
|
||||||
dns.HandleFunc(".", dnsR.handleDnsRequest)
|
|
||||||
|
|
||||||
c.RegisterReloadCallback(func(c *config.C) {
|
|
||||||
reloadDns(l, c)
|
|
||||||
})
|
|
||||||
|
|
||||||
return func() {
|
|
||||||
startDns(l, c)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func getDnsServerAddr(c *config.C) string {
|
func getDnsServerAddr(c *config.C) string {
|
||||||
dnsHost := strings.TrimSpace(c.GetString("lighthouse.dns.host", ""))
|
dnsHost := strings.TrimSpace(c.GetString("lighthouse.dns.host", ""))
|
||||||
// Old guidance was to provide the literal `[::]` in `lighthouse.dns.host` but that won't resolve.
|
// Old guidance was to provide the literal `[::]` in `lighthouse.dns.host` but that won't resolve.
|
||||||
@@ -186,25 +359,3 @@ func getDnsServerAddr(c *config.C) string {
|
|||||||
}
|
}
|
||||||
return net.JoinHostPort(dnsHost, strconv.Itoa(c.GetInt("lighthouse.dns.port", 53)))
|
return net.JoinHostPort(dnsHost, strconv.Itoa(c.GetInt("lighthouse.dns.port", 53)))
|
||||||
}
|
}
|
||||||
|
|
||||||
func startDns(l *logrus.Logger, c *config.C) {
|
|
||||||
dnsAddr = getDnsServerAddr(c)
|
|
||||||
dnsServer = &dns.Server{Addr: dnsAddr, Net: "udp"}
|
|
||||||
l.WithField("dnsListener", dnsAddr).Info("Starting DNS responder")
|
|
||||||
err := dnsServer.ListenAndServe()
|
|
||||||
defer dnsServer.Shutdown()
|
|
||||||
if err != nil {
|
|
||||||
l.Errorf("Failed to start server: %s\n ", err.Error())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func reloadDns(l *logrus.Logger, c *config.C) {
|
|
||||||
if dnsAddr == getDnsServerAddr(c) {
|
|
||||||
l.Debug("No DNS server config change detected")
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
l.Debug("Restarting DNS server")
|
|
||||||
dnsServer.Shutdown()
|
|
||||||
go startDns(l, c)
|
|
||||||
}
|
|
||||||
|
|||||||
+219
-1
@@ -1,19 +1,31 @@
|
|||||||
package nebula
|
package nebula
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
|
"io"
|
||||||
|
"net"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
|
"strconv"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
"github.com/miekg/dns"
|
"github.com/miekg/dns"
|
||||||
"github.com/sirupsen/logrus"
|
"github.com/sirupsen/logrus"
|
||||||
"github.com/slackhq/nebula/config"
|
"github.com/slackhq/nebula/config"
|
||||||
"github.com/stretchr/testify/assert"
|
"github.com/stretchr/testify/assert"
|
||||||
|
"github.com/stretchr/testify/require"
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestParsequery(t *testing.T) {
|
func TestParsequery(t *testing.T) {
|
||||||
l := logrus.New()
|
l := logrus.New()
|
||||||
hostMap := &HostMap{}
|
hostMap := &HostMap{}
|
||||||
ds := newDnsRecords(l, &CertState{}, hostMap)
|
ds := &dnsServer{
|
||||||
|
l: l,
|
||||||
|
dnsMap4: make(map[string]netip.Addr),
|
||||||
|
dnsMap6: make(map[string]netip.Addr),
|
||||||
|
hostMap: hostMap,
|
||||||
|
}
|
||||||
|
ds.enabled.Store(true)
|
||||||
addrs := []netip.Addr{
|
addrs := []netip.Addr{
|
||||||
netip.MustParseAddr("1.2.3.4"),
|
netip.MustParseAddr("1.2.3.4"),
|
||||||
netip.MustParseAddr("1.2.3.5"),
|
netip.MustParseAddr("1.2.3.5"),
|
||||||
@@ -71,3 +83,209 @@ func Test_getDnsServerAddr(t *testing.T) {
|
|||||||
}
|
}
|
||||||
assert.Equal(t, "[::]:1", getDnsServerAddr(c))
|
assert.Equal(t, "[::]:1", getDnsServerAddr(c))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func newTestDnsServer(t *testing.T) (*dnsServer, *config.C) {
|
||||||
|
t.Helper()
|
||||||
|
l := logrus.New()
|
||||||
|
l.Out = io.Discard
|
||||||
|
ds := &dnsServer{
|
||||||
|
l: l,
|
||||||
|
ctx: context.Background(),
|
||||||
|
dnsMap4: make(map[string]netip.Addr),
|
||||||
|
dnsMap6: make(map[string]netip.Addr),
|
||||||
|
hostMap: &HostMap{},
|
||||||
|
}
|
||||||
|
ds.mux = dns.NewServeMux()
|
||||||
|
ds.mux.HandleFunc(".", ds.handleDnsRequest)
|
||||||
|
return ds, config.NewC(l)
|
||||||
|
}
|
||||||
|
|
||||||
|
func setDnsConfig(c *config.C, host string, port string, amLighthouse, serveDns bool) {
|
||||||
|
c.Settings["lighthouse"] = map[string]any{
|
||||||
|
"am_lighthouse": amLighthouse,
|
||||||
|
"serve_dns": serveDns,
|
||||||
|
"dns": map[string]any{
|
||||||
|
"host": host,
|
||||||
|
"port": port,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDnsServer_reload_initial_disabled(t *testing.T) {
|
||||||
|
ds, c := newTestDnsServer(t)
|
||||||
|
setDnsConfig(c, "127.0.0.1", "0", true, false)
|
||||||
|
|
||||||
|
require.NoError(t, ds.reload(c, true))
|
||||||
|
assert.False(t, ds.enabled.Load())
|
||||||
|
assert.Equal(t, "127.0.0.1:0", ds.addr)
|
||||||
|
assert.Nil(t, ds.server)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDnsServer_reload_initial_enabled(t *testing.T) {
|
||||||
|
ds, c := newTestDnsServer(t)
|
||||||
|
setDnsConfig(c, "127.0.0.1", "0", true, true)
|
||||||
|
|
||||||
|
require.NoError(t, ds.reload(c, true))
|
||||||
|
assert.True(t, ds.enabled.Load())
|
||||||
|
assert.Equal(t, "127.0.0.1:0", ds.addr)
|
||||||
|
// initial never starts a runner; that's Control.Start's job
|
||||||
|
assert.Nil(t, ds.server)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDnsServer_reload_initial_serveDnsWithoutLighthouse(t *testing.T) {
|
||||||
|
ds, c := newTestDnsServer(t)
|
||||||
|
setDnsConfig(c, "127.0.0.1", "0", false, true)
|
||||||
|
|
||||||
|
require.NoError(t, ds.reload(c, true))
|
||||||
|
// Wants DNS but isn't a lighthouse: gated off, no runner.
|
||||||
|
assert.False(t, ds.enabled.Load())
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDnsServer_reload_sameAddr_noOp(t *testing.T) {
|
||||||
|
ds, c := newTestDnsServer(t)
|
||||||
|
setDnsConfig(c, "127.0.0.1", "0", true, true)
|
||||||
|
|
||||||
|
require.NoError(t, ds.reload(c, true))
|
||||||
|
// No server running yet, no addr change. Reload should not spawn anything.
|
||||||
|
require.NoError(t, ds.reload(c, false))
|
||||||
|
assert.True(t, ds.enabled.Load())
|
||||||
|
assert.Nil(t, ds.server)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDnsServer_StartStop_lifecycle(t *testing.T) {
|
||||||
|
// Bind to a real (random) UDP port so we exercise the actual
|
||||||
|
// ListenAndServe + Shutdown plumbing including the started-chan race fix.
|
||||||
|
port := freeUDPPort(t)
|
||||||
|
|
||||||
|
ds, c := newTestDnsServer(t)
|
||||||
|
setDnsConfig(c, "127.0.0.1", port, true, true)
|
||||||
|
require.NoError(t, ds.reload(c, true))
|
||||||
|
|
||||||
|
done := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
ds.Start()
|
||||||
|
close(done)
|
||||||
|
}()
|
||||||
|
|
||||||
|
waitFor(t, func() bool {
|
||||||
|
ds.serverMu.Lock()
|
||||||
|
started := ds.started
|
||||||
|
ds.serverMu.Unlock()
|
||||||
|
if started == nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-started:
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
ds.Stop()
|
||||||
|
select {
|
||||||
|
case <-done:
|
||||||
|
case <-time.After(5 * time.Second):
|
||||||
|
t.Fatal("Start did not return after Stop")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDnsServer_Stop_beforeBind_doesNotHang(t *testing.T) {
|
||||||
|
// Stop called immediately after Start should not deadlock even if bind
|
||||||
|
// hasn't completed yet. This exercises the started-chan close-on-bind-fail
|
||||||
|
// path: by binding to an obviously bad port (privileged) we get a fast
|
||||||
|
// bind error before NotifyStartedFunc fires.
|
||||||
|
ds, c := newTestDnsServer(t)
|
||||||
|
// Use a port that should fail to bind (negative would be invalid, use a
|
||||||
|
// host that won't resolve to ensure listenUDP fails quickly).
|
||||||
|
setDnsConfig(c, "256.256.256.256", "53", true, true)
|
||||||
|
require.NoError(t, ds.reload(c, true))
|
||||||
|
|
||||||
|
done := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
ds.Start()
|
||||||
|
close(done)
|
||||||
|
}()
|
||||||
|
|
||||||
|
// Give Start a moment to attempt the bind and fail.
|
||||||
|
select {
|
||||||
|
case <-done:
|
||||||
|
// Bind failed and Start returned; Stop should be a no-op.
|
||||||
|
case <-time.After(time.Second):
|
||||||
|
t.Fatal("Start did not return after a bad bind")
|
||||||
|
}
|
||||||
|
|
||||||
|
stopped := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
ds.Stop()
|
||||||
|
close(stopped)
|
||||||
|
}()
|
||||||
|
select {
|
||||||
|
case <-stopped:
|
||||||
|
case <-time.After(time.Second):
|
||||||
|
t.Fatal("Stop hung after a failed bind")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDnsServer_reload_disable_stopsRunningServer(t *testing.T) {
|
||||||
|
port := freeUDPPort(t)
|
||||||
|
ds, c := newTestDnsServer(t)
|
||||||
|
setDnsConfig(c, "127.0.0.1", port, true, true)
|
||||||
|
require.NoError(t, ds.reload(c, true))
|
||||||
|
|
||||||
|
startReturned := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
ds.Start()
|
||||||
|
close(startReturned)
|
||||||
|
}()
|
||||||
|
waitForBind(t, ds)
|
||||||
|
|
||||||
|
// Toggle serve_dns off; reload should shut the running server down.
|
||||||
|
setDnsConfig(c, "127.0.0.1", port, true, false)
|
||||||
|
require.NoError(t, ds.reload(c, false))
|
||||||
|
select {
|
||||||
|
case <-startReturned:
|
||||||
|
case <-time.After(5 * time.Second):
|
||||||
|
t.Fatal("Start did not return after reload disabled DNS")
|
||||||
|
}
|
||||||
|
assert.False(t, ds.enabled.Load())
|
||||||
|
}
|
||||||
|
|
||||||
|
func freeUDPPort(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
conn, err := net.ListenPacket("udp", "127.0.0.1:0")
|
||||||
|
require.NoError(t, err)
|
||||||
|
port := conn.LocalAddr().(*net.UDPAddr).Port
|
||||||
|
require.NoError(t, conn.Close())
|
||||||
|
return strconv.Itoa(port)
|
||||||
|
}
|
||||||
|
|
||||||
|
func waitForBind(t *testing.T, ds *dnsServer) {
|
||||||
|
t.Helper()
|
||||||
|
waitFor(t, func() bool {
|
||||||
|
ds.serverMu.Lock()
|
||||||
|
started := ds.started
|
||||||
|
ds.serverMu.Unlock()
|
||||||
|
if started == nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-started:
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func waitFor(t *testing.T, cond func() bool) {
|
||||||
|
t.Helper()
|
||||||
|
deadline := time.Now().Add(5 * time.Second)
|
||||||
|
for time.Now().Before(deadline) {
|
||||||
|
if cond() {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
time.Sleep(5 * time.Millisecond)
|
||||||
|
}
|
||||||
|
t.Fatal("timed out waiting for condition")
|
||||||
|
}
|
||||||
|
|||||||
+88
-5
@@ -20,6 +20,7 @@ import (
|
|||||||
"github.com/slackhq/nebula/cert"
|
"github.com/slackhq/nebula/cert"
|
||||||
"github.com/slackhq/nebula/config"
|
"github.com/slackhq/nebula/config"
|
||||||
"github.com/slackhq/nebula/firewall"
|
"github.com/slackhq/nebula/firewall"
|
||||||
|
"github.com/slackhq/nebula/firewall/events"
|
||||||
)
|
)
|
||||||
|
|
||||||
type FirewallInterface interface {
|
type FirewallInterface interface {
|
||||||
@@ -67,6 +68,14 @@ type Firewall struct {
|
|||||||
incomingMetrics firewallMetrics
|
incomingMetrics firewallMetrics
|
||||||
outgoingMetrics firewallMetrics
|
outgoingMetrics firewallMetrics
|
||||||
|
|
||||||
|
// reporter is the optional embedder-supplied event sink. Immutable for
|
||||||
|
// the lifetime of this Firewall; Control.SetFirewallEventReporter
|
||||||
|
// installs it by shallow-copying the Firewall under the conntrack lock
|
||||||
|
// and swapping the pointer, and reloadFirewall carries it forward.
|
||||||
|
// Read unsynchronized on the data path: the preceding Firewall-pointer
|
||||||
|
// read pins the field's value for the duration of that call.
|
||||||
|
reporter events.Reporter
|
||||||
|
|
||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -416,23 +425,27 @@ var ErrNoMatchingRule = errors.New("no matching rule in firewall table")
|
|||||||
|
|
||||||
// Drop returns an error if the packet should be dropped, explaining why. It
|
// Drop returns an error if the packet should be dropped, explaining why. It
|
||||||
// returns nil if the packet should not be dropped.
|
// returns nil if the packet should not be dropped.
|
||||||
func (f *Firewall) Drop(fp firewall.Packet, incoming bool, h *HostInfo, caPool *cert.CAPool, localCache firewall.ConntrackCache) error {
|
func (f *Firewall) Drop(fp firewall.Packet, ctx firewall.PacketContext, incoming bool, h *HostInfo, caPool *cert.CAPool, localCache firewall.ConntrackCache) error {
|
||||||
// Check if we spoke to this tuple, if we did then allow this packet
|
// Check if we spoke to this tuple, if we did then allow this packet
|
||||||
if f.inConns(fp, h, caPool, localCache) {
|
if f.inConns(fp, h, caPool, localCache) {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
peerCert := h.ConnectionState.peerCert
|
||||||
|
|
||||||
// Make sure remote address matches nebula certificate, and determine how to treat it
|
// Make sure remote address matches nebula certificate, and determine how to treat it
|
||||||
if h.networks == nil {
|
if h.networks == nil {
|
||||||
// Simple case: Certificate has one address and no unsafe networks
|
// Simple case: Certificate has one address and no unsafe networks
|
||||||
if h.vpnAddrs[0] != fp.RemoteAddr {
|
if h.vpnAddrs[0] != fp.RemoteAddr {
|
||||||
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
||||||
|
f.reportDrop(incoming, events.DropInvalidRemoteIP, fp, ctx, peerCert)
|
||||||
return ErrInvalidRemoteIP
|
return ErrInvalidRemoteIP
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
nwType, ok := h.networks.Lookup(fp.RemoteAddr)
|
nwType, ok := h.networks.Lookup(fp.RemoteAddr)
|
||||||
if !ok {
|
if !ok {
|
||||||
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
||||||
|
f.reportDrop(incoming, events.DropInvalidRemoteIP, fp, ctx, peerCert)
|
||||||
return ErrInvalidRemoteIP
|
return ErrInvalidRemoteIP
|
||||||
}
|
}
|
||||||
switch nwType {
|
switch nwType {
|
||||||
@@ -440,11 +453,13 @@ func (f *Firewall) Drop(fp firewall.Packet, incoming bool, h *HostInfo, caPool *
|
|||||||
break // nothing special
|
break // nothing special
|
||||||
case NetworkTypeVPNPeer:
|
case NetworkTypeVPNPeer:
|
||||||
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
||||||
|
f.reportDrop(incoming, events.DropPeerRejected, fp, ctx, peerCert)
|
||||||
return ErrPeerRejected // reject for now, one day this may have different FW rules
|
return ErrPeerRejected // reject for now, one day this may have different FW rules
|
||||||
case NetworkTypeUnsafe:
|
case NetworkTypeUnsafe:
|
||||||
break // nothing special, one day this may have different FW rules
|
break // nothing special, one day this may have different FW rules
|
||||||
default:
|
default:
|
||||||
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
f.metrics(incoming).droppedRemoteAddr.Inc(1)
|
||||||
|
f.reportDrop(incoming, events.DropUnknownNetwork, fp, ctx, peerCert)
|
||||||
return ErrUnknownNetworkType //should never happen
|
return ErrUnknownNetworkType //should never happen
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -452,6 +467,7 @@ func (f *Firewall) Drop(fp firewall.Packet, incoming bool, h *HostInfo, caPool *
|
|||||||
// Make sure we are supposed to be handling this local ip address
|
// Make sure we are supposed to be handling this local ip address
|
||||||
if !f.routableNetworks.Contains(fp.LocalAddr) {
|
if !f.routableNetworks.Contains(fp.LocalAddr) {
|
||||||
f.metrics(incoming).droppedLocalAddr.Inc(1)
|
f.metrics(incoming).droppedLocalAddr.Inc(1)
|
||||||
|
f.reportDrop(incoming, events.DropInvalidLocalIP, fp, ctx, peerCert)
|
||||||
return ErrInvalidLocalIP
|
return ErrInvalidLocalIP
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -461,13 +477,14 @@ func (f *Firewall) Drop(fp firewall.Packet, incoming bool, h *HostInfo, caPool *
|
|||||||
}
|
}
|
||||||
|
|
||||||
// We now know which firewall table to check against
|
// We now know which firewall table to check against
|
||||||
if !table.match(fp, incoming, h.ConnectionState.peerCert, caPool) {
|
if !table.match(fp, incoming, peerCert, caPool) {
|
||||||
f.metrics(incoming).droppedNoRule.Inc(1)
|
f.metrics(incoming).droppedNoRule.Inc(1)
|
||||||
|
f.reportDrop(incoming, events.DropNoMatchingRule, fp, ctx, peerCert)
|
||||||
return ErrNoMatchingRule
|
return ErrNoMatchingRule
|
||||||
}
|
}
|
||||||
|
|
||||||
// We always want to conntrack since it is a faster operation
|
// We always want to conntrack since it is a faster operation
|
||||||
f.addConn(fp, incoming)
|
f.addConn(fp, ctx, incoming, peerCert)
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -486,6 +503,59 @@ func (f *Firewall) Destroy() {
|
|||||||
//TODO: clean references if/when needed
|
//TODO: clean references if/when needed
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (f *Firewall) reportDrop(incoming bool, reason events.DropReason, fp firewall.Packet, ctx firewall.PacketContext, peerCert *cert.CachedCertificate) {
|
||||||
|
r := f.reporter
|
||||||
|
if r == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
r.ReportDrop(events.DropEvent{
|
||||||
|
Incoming: incoming,
|
||||||
|
Reason: reason,
|
||||||
|
Packet: fp,
|
||||||
|
Context: ctx,
|
||||||
|
PeerCert: peerCert,
|
||||||
|
RulesVersion: f.rulesVersion,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *Firewall) reportFlowCreate(incoming bool, fp firewall.Packet, ctx firewall.PacketContext, peerCert *cert.CachedCertificate) {
|
||||||
|
r := f.reporter
|
||||||
|
if r == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
r.ReportFlowCreate(events.FlowCreateEvent{
|
||||||
|
Incoming: incoming,
|
||||||
|
Packet: fp,
|
||||||
|
Context: ctx,
|
||||||
|
PeerCert: peerCert,
|
||||||
|
RulesVersion: f.rulesVersion,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *Firewall) reportFlowEvict(incoming bool, fp firewall.Packet, rulesVersion uint16, expired bool) {
|
||||||
|
r := f.reporter
|
||||||
|
if r == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
r.ReportFlowEvict(events.FlowEvictEvent{
|
||||||
|
Incoming: incoming,
|
||||||
|
Packet: fp,
|
||||||
|
RulesVersion: rulesVersion,
|
||||||
|
Expired: expired,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *Firewall) reportRulesReload(oldVersion, newVersion uint16) {
|
||||||
|
r := f.reporter
|
||||||
|
if r == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
r.ReportRulesReload(events.RulesReloadEvent{
|
||||||
|
OldVersion: oldVersion,
|
||||||
|
NewVersion: newVersion,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
func (f *Firewall) EmitStats() {
|
func (f *Firewall) EmitStats() {
|
||||||
conntrack := f.Conntrack
|
conntrack := f.Conntrack
|
||||||
conntrack.Lock()
|
conntrack.Lock()
|
||||||
@@ -536,7 +606,9 @@ func (f *Firewall) inConns(fp firewall.Packet, h *HostInfo, caPool *cert.CAPool,
|
|||||||
WithField("oldRulesVersion", c.rulesVersion).
|
WithField("oldRulesVersion", c.rulesVersion).
|
||||||
Debugln("dropping old conntrack entry, does not match new ruleset")
|
Debugln("dropping old conntrack entry, does not match new ruleset")
|
||||||
}
|
}
|
||||||
|
oldRulesVersion := c.rulesVersion
|
||||||
delete(conntrack.Conns, fp)
|
delete(conntrack.Conns, fp)
|
||||||
|
f.reportFlowEvict(c.incoming, fp, oldRulesVersion, false)
|
||||||
conntrack.Unlock()
|
conntrack.Unlock()
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
@@ -571,7 +643,7 @@ func (f *Firewall) inConns(fp firewall.Packet, h *HostInfo, caPool *cert.CAPool,
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *Firewall) addConn(fp firewall.Packet, incoming bool) {
|
func (f *Firewall) addConn(fp firewall.Packet, ctx firewall.PacketContext, incoming bool, peerCert *cert.CachedCertificate) {
|
||||||
var timeout time.Duration
|
var timeout time.Duration
|
||||||
c := &conn{}
|
c := &conn{}
|
||||||
|
|
||||||
@@ -586,7 +658,8 @@ func (f *Firewall) addConn(fp firewall.Packet, incoming bool) {
|
|||||||
|
|
||||||
conntrack := f.Conntrack
|
conntrack := f.Conntrack
|
||||||
conntrack.Lock()
|
conntrack.Lock()
|
||||||
if _, ok := conntrack.Conns[fp]; !ok {
|
_, existing := conntrack.Conns[fp]
|
||||||
|
if !existing {
|
||||||
conntrack.TimerWheel.Advance(time.Now())
|
conntrack.TimerWheel.Advance(time.Now())
|
||||||
conntrack.TimerWheel.Add(fp, timeout)
|
conntrack.TimerWheel.Add(fp, timeout)
|
||||||
}
|
}
|
||||||
@@ -597,6 +670,13 @@ func (f *Firewall) addConn(fp firewall.Packet, incoming bool) {
|
|||||||
c.rulesVersion = f.rulesVersion
|
c.rulesVersion = f.rulesVersion
|
||||||
c.Expires = time.Now().Add(timeout)
|
c.Expires = time.Now().Add(timeout)
|
||||||
conntrack.Conns[fp] = c
|
conntrack.Conns[fp] = c
|
||||||
|
|
||||||
|
// Report only when this represents a genuinely new flow. Fires under the
|
||||||
|
// conntrack lock so FlowCreate/FlowEvict events stay ordered relative to
|
||||||
|
// RulesReloadEvent, which also fires under this lock.
|
||||||
|
if !existing {
|
||||||
|
f.reportFlowCreate(incoming, fp, ctx, peerCert)
|
||||||
|
}
|
||||||
conntrack.Unlock()
|
conntrack.Unlock()
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -620,7 +700,10 @@ func (f *Firewall) evict(p firewall.Packet) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// This conn is done
|
// This conn is done
|
||||||
|
rulesVersion := t.rulesVersion
|
||||||
|
incoming := t.incoming
|
||||||
delete(conntrack.Conns, p)
|
delete(conntrack.Conns, p)
|
||||||
|
f.reportFlowEvict(incoming, p, rulesVersion, true)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (ft *FirewallTable) match(p firewall.Packet, incoming bool, c *cert.CachedCertificate, caPool *cert.CAPool) bool {
|
func (ft *FirewallTable) match(p firewall.Packet, incoming bool, c *cert.CachedCertificate, caPool *cert.CAPool) bool {
|
||||||
|
|||||||
+12
-5
@@ -1,6 +1,7 @@
|
|||||||
package firewall
|
package firewall
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
@@ -18,7 +19,7 @@ type ConntrackCacheTicker struct {
|
|||||||
cache ConntrackCache
|
cache ConntrackCache
|
||||||
}
|
}
|
||||||
|
|
||||||
func NewConntrackCacheTicker(d time.Duration) *ConntrackCacheTicker {
|
func NewConntrackCacheTicker(ctx context.Context, d time.Duration) *ConntrackCacheTicker {
|
||||||
if d == 0 {
|
if d == 0 {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -27,15 +28,21 @@ func NewConntrackCacheTicker(d time.Duration) *ConntrackCacheTicker {
|
|||||||
cache: ConntrackCache{},
|
cache: ConntrackCache{},
|
||||||
}
|
}
|
||||||
|
|
||||||
go c.tick(d)
|
go c.tick(ctx, d)
|
||||||
|
|
||||||
return c
|
return c
|
||||||
}
|
}
|
||||||
|
|
||||||
func (c *ConntrackCacheTicker) tick(d time.Duration) {
|
func (c *ConntrackCacheTicker) tick(ctx context.Context, d time.Duration) {
|
||||||
|
t := time.NewTicker(d)
|
||||||
|
defer t.Stop()
|
||||||
for {
|
for {
|
||||||
time.Sleep(d)
|
select {
|
||||||
c.cacheTick.Add(1)
|
case <-ctx.Done():
|
||||||
|
return
|
||||||
|
case <-t.C:
|
||||||
|
c.cacheTick.Add(1)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,102 @@
|
|||||||
|
// Package events defines the opt-in firewall event reporting interface.
|
||||||
|
//
|
||||||
|
// Nebula emits raw packet-level events (drops, flow creations, flow evictions,
|
||||||
|
// rule reloads) and does no aggregation, counting, batching, rule-description,
|
||||||
|
// transport, or timestamping. Embedders correlate events back to yaml rules
|
||||||
|
// out of band and capture whatever clock they need themselves. All Report*
|
||||||
|
// methods are invoked while nebula holds internal locks and must be
|
||||||
|
// non-blocking.
|
||||||
|
//
|
||||||
|
// Events are passed to Report* methods by value. Implementations must not
|
||||||
|
// take the address of a received event: doing so forces Go's escape
|
||||||
|
// analysis to move the event to the heap and costs one allocation per call.
|
||||||
|
// To forward an event, either copy its fields into the reporter's own
|
||||||
|
// pooled record or send it through a value-typed channel (chan DropEvent,
|
||||||
|
// not chan *DropEvent).
|
||||||
|
package events
|
||||||
|
|
||||||
|
import (
|
||||||
|
"github.com/slackhq/nebula/cert"
|
||||||
|
"github.com/slackhq/nebula/firewall"
|
||||||
|
)
|
||||||
|
|
||||||
|
type DropReason uint8
|
||||||
|
|
||||||
|
const (
|
||||||
|
DropInvalidLocalIP DropReason = iota
|
||||||
|
DropInvalidRemoteIP
|
||||||
|
DropPeerRejected
|
||||||
|
DropUnknownNetwork
|
||||||
|
DropNoMatchingRule
|
||||||
|
)
|
||||||
|
|
||||||
|
func (r DropReason) String() string {
|
||||||
|
switch r {
|
||||||
|
case DropInvalidLocalIP:
|
||||||
|
return "invalid_local_ip"
|
||||||
|
case DropInvalidRemoteIP:
|
||||||
|
return "invalid_remote_ip"
|
||||||
|
case DropPeerRejected:
|
||||||
|
return "peer_rejected"
|
||||||
|
case DropUnknownNetwork:
|
||||||
|
return "unknown_network"
|
||||||
|
case DropNoMatchingRule:
|
||||||
|
return "no_matching_rule"
|
||||||
|
default:
|
||||||
|
return "unknown"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// DropEvent is emitted for every packet that fails the firewall check. Drops
|
||||||
|
// are not aggregated; every drop produces one event.
|
||||||
|
type DropEvent struct {
|
||||||
|
Incoming bool
|
||||||
|
Reason DropReason
|
||||||
|
Packet firewall.Packet
|
||||||
|
Context firewall.PacketContext
|
||||||
|
PeerCert *cert.CachedCertificate
|
||||||
|
RulesVersion uint16
|
||||||
|
}
|
||||||
|
|
||||||
|
// FlowCreateEvent is emitted when a packet is allowed and a new conntrack
|
||||||
|
// entry is created. Subsequent packets in the same flow do not re-emit.
|
||||||
|
type FlowCreateEvent struct {
|
||||||
|
Incoming bool
|
||||||
|
Packet firewall.Packet
|
||||||
|
Context firewall.PacketContext
|
||||||
|
PeerCert *cert.CachedCertificate
|
||||||
|
RulesVersion uint16
|
||||||
|
}
|
||||||
|
|
||||||
|
// FlowEvictEvent is emitted when a conntrack entry is removed. Context is
|
||||||
|
// not carried: timer-wheel eviction has no packet in hand, and reload
|
||||||
|
// revalidation evicts the OLD flow rather than the triggering packet.
|
||||||
|
// RulesVersion is the version under which the flow was originally allowed,
|
||||||
|
// which may differ from the current firewall version.
|
||||||
|
type FlowEvictEvent struct {
|
||||||
|
Incoming bool
|
||||||
|
Packet firewall.Packet
|
||||||
|
RulesVersion uint16
|
||||||
|
// Expired is true when eviction was due to conntrack timeout; false when
|
||||||
|
// the entry was removed because it failed re-validation after a reload.
|
||||||
|
Expired bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// RulesReloadEvent is emitted once after each successful firewall reload.
|
||||||
|
// Reporters that bucket state by RulesVersion should close the old bucket
|
||||||
|
// and open a new one on receipt.
|
||||||
|
type RulesReloadEvent struct {
|
||||||
|
OldVersion uint16
|
||||||
|
NewVersion uint16
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reporter is the embedder-supplied sink for firewall events. Implementations
|
||||||
|
// that want a timestamp should call time.Now() themselves at the top of the
|
||||||
|
// method; nebula does not provide one. See the package doc for the
|
||||||
|
// do-not-take-address rule.
|
||||||
|
type Reporter interface {
|
||||||
|
ReportDrop(DropEvent)
|
||||||
|
ReportFlowCreate(FlowCreateEvent)
|
||||||
|
ReportFlowEvict(FlowEvictEvent)
|
||||||
|
ReportRulesReload(RulesReloadEvent)
|
||||||
|
}
|
||||||
@@ -31,6 +31,27 @@ type Packet struct {
|
|||||||
Fragment bool
|
Fragment bool
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// PacketContext carries additional parsed details about a packet that are
|
||||||
|
// useful for event reporting but deliberately kept out of Packet so Packet
|
||||||
|
// can keep being used as a conntrack map key. Populated alongside Packet by
|
||||||
|
// newPacket.
|
||||||
|
//
|
||||||
|
// Fields are interpreted based on Packet.Protocol:
|
||||||
|
// - ProtoTCP: TCPFlags is meaningful; ICMPType / ICMPCode are zero
|
||||||
|
// - ProtoICMP, ProtoICMPv6: ICMPType / ICMPCode are meaningful; TCPFlags is zero
|
||||||
|
// - ProtoUDP and others: only Length is meaningful
|
||||||
|
type PacketContext struct {
|
||||||
|
// Length is the total IP packet length in bytes, including headers.
|
||||||
|
Length uint16
|
||||||
|
// TCPFlags is the flag byte from the TCP header (bits for FIN, SYN, RST,
|
||||||
|
// PSH, ACK, URG, ECE, CWR).
|
||||||
|
TCPFlags uint8
|
||||||
|
// ICMPType is the type field of the ICMP / ICMPv6 header.
|
||||||
|
ICMPType uint8
|
||||||
|
// ICMPCode is the code field of the ICMP / ICMPv6 header.
|
||||||
|
ICMPCode uint8
|
||||||
|
}
|
||||||
|
|
||||||
func (fp *Packet) Copy() *Packet {
|
func (fp *Packet) Copy() *Packet {
|
||||||
return &Packet{
|
return &Packet{
|
||||||
LocalAddr: fp.LocalAddr,
|
LocalAddr: fp.LocalAddr,
|
||||||
|
|||||||
@@ -0,0 +1,731 @@
|
|||||||
|
package nebula
|
||||||
|
|
||||||
|
import (
|
||||||
|
"net"
|
||||||
|
"net/netip"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/gaissmai/bart"
|
||||||
|
"github.com/google/gopacket"
|
||||||
|
"github.com/google/gopacket/layers"
|
||||||
|
"github.com/slackhq/nebula/cert"
|
||||||
|
"github.com/slackhq/nebula/firewall"
|
||||||
|
"github.com/slackhq/nebula/firewall/events"
|
||||||
|
"github.com/slackhq/nebula/test"
|
||||||
|
"github.com/stretchr/testify/assert"
|
||||||
|
"github.com/stretchr/testify/require"
|
||||||
|
)
|
||||||
|
|
||||||
|
// recordingReporter captures every event fired against it. Its methods take
|
||||||
|
// the conntrack lock implicitly (via the firewall code path that invokes
|
||||||
|
// them), so we synchronize accumulator mutations with a small mutex to keep
|
||||||
|
// the race detector happy across goroutines in case a test introduces any.
|
||||||
|
type recordingReporter struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
drops []recordedDrop
|
||||||
|
creates []recordedCreate
|
||||||
|
evicts []recordedEvict
|
||||||
|
reloads []recordedReload
|
||||||
|
}
|
||||||
|
|
||||||
|
type recordedDrop struct {
|
||||||
|
incoming bool
|
||||||
|
reason events.DropReason
|
||||||
|
remote netip.Addr
|
||||||
|
local netip.Addr
|
||||||
|
peerName string
|
||||||
|
rulesVersion uint16
|
||||||
|
ctx firewall.PacketContext
|
||||||
|
}
|
||||||
|
|
||||||
|
type recordedCreate struct {
|
||||||
|
incoming bool
|
||||||
|
remote netip.Addr
|
||||||
|
local netip.Addr
|
||||||
|
peerName string
|
||||||
|
rulesVersion uint16
|
||||||
|
ctx firewall.PacketContext
|
||||||
|
}
|
||||||
|
|
||||||
|
type recordedEvict struct {
|
||||||
|
incoming bool
|
||||||
|
remote netip.Addr
|
||||||
|
local netip.Addr
|
||||||
|
rulesVersion uint16
|
||||||
|
expired bool
|
||||||
|
}
|
||||||
|
|
||||||
|
type recordedReload struct {
|
||||||
|
oldVersion uint16
|
||||||
|
newVersion uint16
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *recordingReporter) ReportDrop(e events.DropEvent) {
|
||||||
|
r.mu.Lock()
|
||||||
|
defer r.mu.Unlock()
|
||||||
|
name := ""
|
||||||
|
if e.PeerCert != nil && e.PeerCert.Certificate != nil {
|
||||||
|
name = e.PeerCert.Certificate.Name()
|
||||||
|
}
|
||||||
|
r.drops = append(r.drops, recordedDrop{
|
||||||
|
incoming: e.Incoming,
|
||||||
|
reason: e.Reason,
|
||||||
|
remote: e.Packet.RemoteAddr,
|
||||||
|
local: e.Packet.LocalAddr,
|
||||||
|
peerName: name,
|
||||||
|
rulesVersion: e.RulesVersion,
|
||||||
|
ctx: e.Context,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *recordingReporter) ReportFlowCreate(e events.FlowCreateEvent) {
|
||||||
|
r.mu.Lock()
|
||||||
|
defer r.mu.Unlock()
|
||||||
|
name := ""
|
||||||
|
if e.PeerCert != nil && e.PeerCert.Certificate != nil {
|
||||||
|
name = e.PeerCert.Certificate.Name()
|
||||||
|
}
|
||||||
|
r.creates = append(r.creates, recordedCreate{
|
||||||
|
incoming: e.Incoming,
|
||||||
|
remote: e.Packet.RemoteAddr,
|
||||||
|
local: e.Packet.LocalAddr,
|
||||||
|
peerName: name,
|
||||||
|
rulesVersion: e.RulesVersion,
|
||||||
|
ctx: e.Context,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *recordingReporter) ReportFlowEvict(e events.FlowEvictEvent) {
|
||||||
|
r.mu.Lock()
|
||||||
|
defer r.mu.Unlock()
|
||||||
|
r.evicts = append(r.evicts, recordedEvict{
|
||||||
|
incoming: e.Incoming,
|
||||||
|
remote: e.Packet.RemoteAddr,
|
||||||
|
local: e.Packet.LocalAddr,
|
||||||
|
rulesVersion: e.RulesVersion,
|
||||||
|
expired: e.Expired,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *recordingReporter) ReportRulesReload(e events.RulesReloadEvent) {
|
||||||
|
r.mu.Lock()
|
||||||
|
defer r.mu.Unlock()
|
||||||
|
r.reloads = append(r.reloads, recordedReload{
|
||||||
|
oldVersion: e.OldVersion,
|
||||||
|
newVersion: e.NewVersion,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// eventFixture builds a Firewall wired to a Control plus a packet/hostinfo
|
||||||
|
// pair that a test can reuse. By default the ruleset allows the packet;
|
||||||
|
// callers mutate fw / p / h as needed before invoking Drop.
|
||||||
|
type eventFixture struct {
|
||||||
|
ctl *Control
|
||||||
|
fw *Firewall
|
||||||
|
p firewall.Packet
|
||||||
|
h *HostInfo
|
||||||
|
cp *cert.CAPool
|
||||||
|
}
|
||||||
|
|
||||||
|
func newEventFixture(t *testing.T) *eventFixture {
|
||||||
|
t.Helper()
|
||||||
|
l := test.NewLogger()
|
||||||
|
|
||||||
|
// myVpnNetworksTable covers our single peer address so buildNetworks takes
|
||||||
|
// the "simple case" path (h.networks stays nil); tests that want a populated
|
||||||
|
// BART table overwrite h.networks directly.
|
||||||
|
vpnNetworks := new(bart.Lite)
|
||||||
|
vpnNetworks.Insert(netip.MustParsePrefix("1.2.3.0/24"))
|
||||||
|
|
||||||
|
// Use the same cert for "peer" and "local" endpoints, matching the
|
||||||
|
// TestFirewall_Drop fixture style: LocalAddr == RemoteAddr == peer vpn addr.
|
||||||
|
c := &dummyCert{
|
||||||
|
name: "host1",
|
||||||
|
networks: []netip.Prefix{netip.MustParsePrefix("1.2.3.4/24")},
|
||||||
|
groups: []string{"default-group"},
|
||||||
|
issuer: "signer-shasum",
|
||||||
|
}
|
||||||
|
h := &HostInfo{
|
||||||
|
ConnectionState: &ConnectionState{
|
||||||
|
peerCert: &cert.CachedCertificate{
|
||||||
|
Certificate: c,
|
||||||
|
InvertedGroups: map[string]struct{}{"default-group": {}},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
vpnAddrs: []netip.Addr{netip.MustParseAddr("1.2.3.4")},
|
||||||
|
}
|
||||||
|
h.buildNetworks(vpnNetworks, c)
|
||||||
|
|
||||||
|
fw := NewFirewall(l, time.Minute, time.Minute, time.Minute, c)
|
||||||
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"any"}, "", "", "", "", ""))
|
||||||
|
require.NoError(t, fw.AddRule(false, firewall.ProtoAny, 0, 0, []string{"any"}, "", "", "", "", ""))
|
||||||
|
|
||||||
|
ctl := &Control{
|
||||||
|
f: &Interface{firewall: fw},
|
||||||
|
l: l,
|
||||||
|
}
|
||||||
|
|
||||||
|
return &eventFixture{
|
||||||
|
ctl: ctl,
|
||||||
|
fw: fw,
|
||||||
|
p: firewall.Packet{
|
||||||
|
LocalAddr: netip.MustParseAddr("1.2.3.4"),
|
||||||
|
RemoteAddr: netip.MustParseAddr("1.2.3.4"),
|
||||||
|
LocalPort: 10,
|
||||||
|
RemotePort: 90,
|
||||||
|
Protocol: firewall.ProtoUDP,
|
||||||
|
},
|
||||||
|
h: h,
|
||||||
|
cp: cert.NewCAPool(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// firewall() returns the currently-installed firewall. Needed because
|
||||||
|
// SetFirewallEventReporter replaces it via shallow-copy swap.
|
||||||
|
func (f *eventFixture) firewall() *Firewall {
|
||||||
|
return f.ctl.f.firewall
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportDrop_InvalidRemoteIP(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
// Packet to an address not in the cert's networks.
|
||||||
|
f.p.RemoteAddr = netip.MustParseAddr("9.9.9.9")
|
||||||
|
assert.Equal(t, ErrInvalidRemoteIP, f.firewall().Drop(f.p, firewall.PacketContext{}, false, f.h, f.cp, nil))
|
||||||
|
|
||||||
|
require.Len(t, r.drops, 1)
|
||||||
|
assert.Equal(t, events.DropInvalidRemoteIP, r.drops[0].reason)
|
||||||
|
assert.False(t, r.drops[0].incoming)
|
||||||
|
assert.Equal(t, "host1", r.drops[0].peerName)
|
||||||
|
assert.Empty(t, r.creates)
|
||||||
|
assert.Empty(t, r.evicts)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportDrop_InvalidLocalIP(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
// LocalAddr outside our routable networks.
|
||||||
|
f.p.LocalAddr = netip.MustParseAddr("9.9.9.9")
|
||||||
|
assert.Equal(t, ErrInvalidLocalIP, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
|
||||||
|
require.Len(t, r.drops, 1)
|
||||||
|
assert.Equal(t, events.DropInvalidLocalIP, r.drops[0].reason)
|
||||||
|
assert.True(t, r.drops[0].incoming)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportDrop_NoMatchingRule(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
// Reset to a firewall with no matching rule.
|
||||||
|
l := test.NewLogger()
|
||||||
|
fw := NewFirewall(l, time.Minute, time.Minute, time.Minute, f.h.ConnectionState.peerCert.Certificate)
|
||||||
|
// Rule that won't match (group not in peer's groups).
|
||||||
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", ""))
|
||||||
|
require.NoError(t, fw.AddRule(false, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", ""))
|
||||||
|
f.ctl.f.firewall = fw
|
||||||
|
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
assert.Equal(t, ErrNoMatchingRule, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
require.Len(t, r.drops, 1)
|
||||||
|
assert.Equal(t, events.DropNoMatchingRule, r.drops[0].reason)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportDrop_PeerRejected(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
// Re-classify the remote as VPNPeer so it triggers DropPeerRejected.
|
||||||
|
f.h.networks = new(bart.Table[NetworkType])
|
||||||
|
f.h.networks.Insert(netip.MustParsePrefix("1.2.3.0/24"), NetworkTypeVPNPeer)
|
||||||
|
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
assert.Equal(t, ErrPeerRejected, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
require.Len(t, r.drops, 1)
|
||||||
|
assert.Equal(t, events.DropPeerRejected, r.drops[0].reason)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportDrop_UnknownNetwork(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
// Insert an unrecognized NetworkType value to hit the default branch.
|
||||||
|
f.h.networks = new(bart.Table[NetworkType])
|
||||||
|
f.h.networks.Insert(netip.MustParsePrefix("1.2.3.0/24"), NetworkTypeUnknown)
|
||||||
|
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
assert.Equal(t, ErrUnknownNetworkType, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
require.Len(t, r.drops, 1)
|
||||||
|
assert.Equal(t, events.DropUnknownNetwork, r.drops[0].reason)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportFlowCreate_OnceOnly(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
// First allowed packet creates the conntrack entry.
|
||||||
|
require.NoError(t, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
// Second matching packet on the same tuple is short-circuited by conntrack
|
||||||
|
// and must not fire another FlowCreate.
|
||||||
|
require.NoError(t, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
|
||||||
|
require.Len(t, r.creates, 1)
|
||||||
|
assert.True(t, r.creates[0].incoming)
|
||||||
|
assert.Equal(t, f.p.RemoteAddr, r.creates[0].remote)
|
||||||
|
assert.Empty(t, r.drops)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportFlowEvict_OnReloadPurge(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
// Create a flow under the current rules.
|
||||||
|
require.NoError(t, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
require.Len(t, r.creates, 1)
|
||||||
|
|
||||||
|
// Simulate a reload that produces rules the existing flow no longer
|
||||||
|
// matches. Bump rulesVersion and replace InRules with an empty table so
|
||||||
|
// revalidation fails.
|
||||||
|
fw := f.firewall()
|
||||||
|
fw.Conntrack.Lock()
|
||||||
|
fw.rulesVersion++
|
||||||
|
fw.InRules = newFirewallTable()
|
||||||
|
fw.Conntrack.Unlock()
|
||||||
|
|
||||||
|
// Next packet triggers re-validation, which fails and evicts the entry.
|
||||||
|
err := fw.Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil)
|
||||||
|
assert.Equal(t, ErrNoMatchingRule, err)
|
||||||
|
|
||||||
|
require.Len(t, r.evicts, 1)
|
||||||
|
assert.False(t, r.evicts[0].expired, "evict from reload purge is not expiration")
|
||||||
|
assert.True(t, r.evicts[0].incoming)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReportFlowEvict_OnTimeout(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
require.NoError(t, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
require.Len(t, r.creates, 1)
|
||||||
|
|
||||||
|
// Force expiration by rewinding the entry's deadline.
|
||||||
|
fw := f.firewall()
|
||||||
|
fw.Conntrack.Lock()
|
||||||
|
c := fw.Conntrack.Conns[f.p]
|
||||||
|
require.NotNil(t, c)
|
||||||
|
c.Expires = time.Now().Add(-time.Hour)
|
||||||
|
fw.evict(f.p)
|
||||||
|
fw.Conntrack.Unlock()
|
||||||
|
|
||||||
|
require.Len(t, r.evicts, 1)
|
||||||
|
assert.True(t, r.evicts[0].expired)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_SetNil_Clears(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
require.NoError(t, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
require.Len(t, r.creates, 1)
|
||||||
|
|
||||||
|
f.ctl.SetFirewallEventReporter(nil)
|
||||||
|
resetConntrack(f.firewall())
|
||||||
|
require.NoError(t, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
// No second create should be recorded.
|
||||||
|
assert.Len(t, r.creates, 1)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_ReporterSurvivesSwap(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
// Simulate a reload by swapping in a fresh Firewall that carries the
|
||||||
|
// reporter forward. Mirrors what reloadFirewall does with the shared
|
||||||
|
// conntrack pointer.
|
||||||
|
l := test.NewLogger()
|
||||||
|
oldFw := f.firewall()
|
||||||
|
newFw := NewFirewall(l, time.Minute, time.Minute, time.Minute, f.h.ConnectionState.peerCert.Certificate)
|
||||||
|
require.NoError(t, newFw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"any"}, "", "", "", "", ""))
|
||||||
|
require.NoError(t, newFw.AddRule(false, firewall.ProtoAny, 0, 0, []string{"any"}, "", "", "", "", ""))
|
||||||
|
newFw.Conntrack = oldFw.Conntrack
|
||||||
|
newFw.rulesVersion = oldFw.rulesVersion + 1
|
||||||
|
newFw.reporter = oldFw.reporter
|
||||||
|
f.ctl.f.firewall = newFw
|
||||||
|
newFw.reportRulesReload(oldFw.rulesVersion, newFw.rulesVersion)
|
||||||
|
|
||||||
|
require.Len(t, r.reloads, 1)
|
||||||
|
assert.Equal(t, oldFw.rulesVersion, r.reloads[0].oldVersion)
|
||||||
|
assert.Equal(t, newFw.rulesVersion, r.reloads[0].newVersion)
|
||||||
|
|
||||||
|
// Events on the new firewall should still reach the same reporter.
|
||||||
|
require.NoError(t, newFw.Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
require.Len(t, r.creates, 1)
|
||||||
|
assert.Equal(t, newFw.rulesVersion, r.creates[0].rulesVersion)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEvents_InstallDoesNotMutateOldFirewall(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
before := f.firewall()
|
||||||
|
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
after := f.firewall()
|
||||||
|
assert.NotSame(t, before, after, "SetFirewallEventReporter must replace the Firewall pointer")
|
||||||
|
assert.Nil(t, before.reporter, "the pre-install Firewall must remain untouched")
|
||||||
|
assert.NotNil(t, after.reporter)
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- PacketContext parse tests --------------------------------------------
|
||||||
|
|
||||||
|
func mustSerialize(t *testing.T, lrs ...gopacket.SerializableLayer) []byte {
|
||||||
|
t.Helper()
|
||||||
|
buf := gopacket.NewSerializeBuffer()
|
||||||
|
opt := gopacket.SerializeOptions{ComputeChecksums: false, FixLengths: true}
|
||||||
|
require.NoError(t, gopacket.SerializeLayers(buf, opt, lrs...))
|
||||||
|
return buf.Bytes()
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPacketContext_IPv4_TCPFlags(t *testing.T) {
|
||||||
|
ip := &layers.IPv4{
|
||||||
|
Version: 4, TTL: 64, Protocol: layers.IPProtocolTCP,
|
||||||
|
SrcIP: net.IPv4(10, 0, 0, 1), DstIP: net.IPv4(10, 0, 0, 2),
|
||||||
|
}
|
||||||
|
tcp := &layers.TCP{SrcPort: 1234, DstPort: 80, SYN: true, ACK: true}
|
||||||
|
require.NoError(t, tcp.SetNetworkLayerForChecksum(ip))
|
||||||
|
data := mustSerialize(t, ip, tcp, gopacket.Payload([]byte("hello")))
|
||||||
|
|
||||||
|
var fp firewall.Packet
|
||||||
|
var ctx firewall.PacketContext
|
||||||
|
require.NoError(t, newPacket(data, true, &fp, &ctx))
|
||||||
|
|
||||||
|
assert.Equal(t, uint8(firewall.ProtoTCP), fp.Protocol)
|
||||||
|
// SYN (0x02) + ACK (0x10) = 0x12
|
||||||
|
assert.Equal(t, uint8(0x12), ctx.TCPFlags)
|
||||||
|
assert.Equal(t, uint16(len(data)), ctx.Length)
|
||||||
|
assert.Equal(t, uint8(0), ctx.ICMPType)
|
||||||
|
assert.Equal(t, uint8(0), ctx.ICMPCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPacketContext_IPv4_ICMPTypeCode(t *testing.T) {
|
||||||
|
ip := &layers.IPv4{
|
||||||
|
Version: 4, TTL: 64, Protocol: layers.IPProtocolICMPv4,
|
||||||
|
SrcIP: net.IPv4(10, 0, 0, 1), DstIP: net.IPv4(10, 0, 0, 2),
|
||||||
|
}
|
||||||
|
// Destination Unreachable, code 3 (port unreachable)
|
||||||
|
icmp := &layers.ICMPv4{
|
||||||
|
TypeCode: layers.CreateICMPv4TypeCode(layers.ICMPv4TypeDestinationUnreachable, layers.ICMPv4CodePort),
|
||||||
|
}
|
||||||
|
data := mustSerialize(t, ip, icmp, gopacket.Payload([]byte{0, 0, 0, 0}))
|
||||||
|
|
||||||
|
var fp firewall.Packet
|
||||||
|
var ctx firewall.PacketContext
|
||||||
|
require.NoError(t, newPacket(data, true, &fp, &ctx))
|
||||||
|
|
||||||
|
assert.Equal(t, uint8(firewall.ProtoICMP), fp.Protocol)
|
||||||
|
assert.Equal(t, uint8(layers.ICMPv4TypeDestinationUnreachable), ctx.ICMPType)
|
||||||
|
assert.Equal(t, uint8(layers.ICMPv4CodePort), ctx.ICMPCode)
|
||||||
|
assert.Equal(t, uint16(len(data)), ctx.Length)
|
||||||
|
assert.Equal(t, uint8(0), ctx.TCPFlags)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPacketContext_IPv4_UDPLengthOnly(t *testing.T) {
|
||||||
|
ip := &layers.IPv4{
|
||||||
|
Version: 4, TTL: 64, Protocol: layers.IPProtocolUDP,
|
||||||
|
SrcIP: net.IPv4(10, 0, 0, 1), DstIP: net.IPv4(10, 0, 0, 2),
|
||||||
|
}
|
||||||
|
udp := &layers.UDP{SrcPort: 1234, DstPort: 53}
|
||||||
|
require.NoError(t, udp.SetNetworkLayerForChecksum(ip))
|
||||||
|
data := mustSerialize(t, ip, udp, gopacket.Payload([]byte("query")))
|
||||||
|
|
||||||
|
var fp firewall.Packet
|
||||||
|
var ctx firewall.PacketContext
|
||||||
|
require.NoError(t, newPacket(data, true, &fp, &ctx))
|
||||||
|
|
||||||
|
assert.Equal(t, uint8(firewall.ProtoUDP), fp.Protocol)
|
||||||
|
assert.Equal(t, uint16(len(data)), ctx.Length)
|
||||||
|
assert.Zero(t, ctx.TCPFlags)
|
||||||
|
assert.Zero(t, ctx.ICMPType)
|
||||||
|
assert.Zero(t, ctx.ICMPCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPacketContext_IPv6_TCPFlags(t *testing.T) {
|
||||||
|
ip := &layers.IPv6{
|
||||||
|
Version: 6, HopLimit: 64, NextHeader: layers.IPProtocolTCP,
|
||||||
|
SrcIP: net.ParseIP("fd00::1"), DstIP: net.ParseIP("fd00::2"),
|
||||||
|
}
|
||||||
|
tcp := &layers.TCP{SrcPort: 1234, DstPort: 443, FIN: true, ACK: true}
|
||||||
|
require.NoError(t, tcp.SetNetworkLayerForChecksum(ip))
|
||||||
|
data := mustSerialize(t, ip, tcp, gopacket.Payload([]byte("bye")))
|
||||||
|
|
||||||
|
var fp firewall.Packet
|
||||||
|
var ctx firewall.PacketContext
|
||||||
|
require.NoError(t, newPacket(data, true, &fp, &ctx))
|
||||||
|
|
||||||
|
assert.Equal(t, uint8(firewall.ProtoTCP), fp.Protocol)
|
||||||
|
// FIN (0x01) + ACK (0x10) = 0x11
|
||||||
|
assert.Equal(t, uint8(0x11), ctx.TCPFlags)
|
||||||
|
assert.Equal(t, uint16(len(data)), ctx.Length)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPacketContext_IPv6_ICMPv6TypeCode(t *testing.T) {
|
||||||
|
ip := &layers.IPv6{
|
||||||
|
Version: 6, HopLimit: 64, NextHeader: layers.IPProtocolICMPv6,
|
||||||
|
SrcIP: net.ParseIP("fd00::1"), DstIP: net.ParseIP("fd00::2"),
|
||||||
|
}
|
||||||
|
icmp := &layers.ICMPv6{
|
||||||
|
TypeCode: layers.CreateICMPv6TypeCode(layers.ICMPv6TypeDestinationUnreachable, layers.ICMPv6CodePortUnreachable),
|
||||||
|
}
|
||||||
|
require.NoError(t, icmp.SetNetworkLayerForChecksum(ip))
|
||||||
|
data := mustSerialize(t, ip, icmp, gopacket.Payload([]byte{0, 0, 0, 0, 0, 0, 0, 0}))
|
||||||
|
|
||||||
|
var fp firewall.Packet
|
||||||
|
var ctx firewall.PacketContext
|
||||||
|
require.NoError(t, newPacket(data, true, &fp, &ctx))
|
||||||
|
|
||||||
|
assert.Equal(t, uint8(firewall.ProtoICMPv6), fp.Protocol)
|
||||||
|
assert.Equal(t, uint8(layers.ICMPv6TypeDestinationUnreachable), ctx.ICMPType)
|
||||||
|
assert.Equal(t, uint8(layers.ICMPv6CodePortUnreachable), ctx.ICMPCode)
|
||||||
|
assert.Equal(t, uint16(len(data)), ctx.Length)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestPacketContext_NilOK confirms a nil context pointer is accepted by
|
||||||
|
// newPacket (the hot path may elect not to pass one).
|
||||||
|
func TestPacketContext_NilOK(t *testing.T) {
|
||||||
|
ip := &layers.IPv4{
|
||||||
|
Version: 4, TTL: 64, Protocol: layers.IPProtocolUDP,
|
||||||
|
SrcIP: net.IPv4(10, 0, 0, 1), DstIP: net.IPv4(10, 0, 0, 2),
|
||||||
|
}
|
||||||
|
udp := &layers.UDP{SrcPort: 1, DstPort: 2}
|
||||||
|
require.NoError(t, udp.SetNetworkLayerForChecksum(ip))
|
||||||
|
data := mustSerialize(t, ip, udp)
|
||||||
|
|
||||||
|
var fp firewall.Packet
|
||||||
|
require.NoError(t, newPacket(data, true, &fp, nil))
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestPacketContext_FlowCreateCarriesContext exercises the full Drop -> addConn
|
||||||
|
// -> ReportFlowCreate path with a realistic TCP packet and confirms the
|
||||||
|
// context makes it into the reporter.
|
||||||
|
func TestPacketContext_FlowCreateCarriesContext(t *testing.T) {
|
||||||
|
f := newEventFixture(t)
|
||||||
|
r := &recordingReporter{}
|
||||||
|
f.ctl.SetFirewallEventReporter(r)
|
||||||
|
|
||||||
|
// Hand-construct a matching TCP packet.
|
||||||
|
ctx := firewall.PacketContext{Length: 1500, TCPFlags: 0x12}
|
||||||
|
p := f.p
|
||||||
|
p.Protocol = firewall.ProtoTCP
|
||||||
|
require.NoError(t, f.firewall().Drop(p, ctx, true, f.h, f.cp, nil))
|
||||||
|
|
||||||
|
require.Len(t, r.creates, 1)
|
||||||
|
assert.Equal(t, uint16(1500), r.creates[0].ctx.Length)
|
||||||
|
assert.Equal(t, uint8(0x12), r.creates[0].ctx.TCPFlags)
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- benchmarks ------------------------------------------------------------
|
||||||
|
|
||||||
|
// noopReporter is the cheapest possible reporter. Methods discard the event.
|
||||||
|
type noopReporter struct{}
|
||||||
|
|
||||||
|
func (noopReporter) ReportDrop(events.DropEvent) {}
|
||||||
|
func (noopReporter) ReportFlowCreate(events.FlowCreateEvent) {}
|
||||||
|
func (noopReporter) ReportFlowEvict(events.FlowEvictEvent) {}
|
||||||
|
func (noopReporter) ReportRulesReload(events.RulesReloadEvent) {
|
||||||
|
}
|
||||||
|
|
||||||
|
// bufferedReporter demonstrates a realistic zero-alloc reporter: each event
|
||||||
|
// is forwarded to a value-typed channel. The channel send is a memcpy into
|
||||||
|
// the channel's pre-allocated ring buffer -- no heap traffic. A background
|
||||||
|
// goroutine would drain these; the bench skips draining to keep the report
|
||||||
|
// path pure.
|
||||||
|
type bufferedReporter struct {
|
||||||
|
drops chan events.DropEvent
|
||||||
|
flows chan events.FlowCreateEvent
|
||||||
|
evicts chan events.FlowEvictEvent
|
||||||
|
}
|
||||||
|
|
||||||
|
func newBufferedReporter(cap int) *bufferedReporter {
|
||||||
|
return &bufferedReporter{
|
||||||
|
drops: make(chan events.DropEvent, cap),
|
||||||
|
flows: make(chan events.FlowCreateEvent, cap),
|
||||||
|
evicts: make(chan events.FlowEvictEvent, cap),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *bufferedReporter) ReportDrop(e events.DropEvent) {
|
||||||
|
select {
|
||||||
|
case r.drops <- e:
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *bufferedReporter) ReportFlowCreate(e events.FlowCreateEvent) {
|
||||||
|
select {
|
||||||
|
case r.flows <- e:
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *bufferedReporter) ReportFlowEvict(e events.FlowEvictEvent) {
|
||||||
|
select {
|
||||||
|
case r.evicts <- e:
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *bufferedReporter) ReportRulesReload(events.RulesReloadEvent) {}
|
||||||
|
|
||||||
|
// pointerReporter is the anti-pattern: it takes the address of the incoming
|
||||||
|
// event struct, which forces the callee-side copy onto the heap. Kept for
|
||||||
|
// comparison so we can see the alloc cost an unwary reporter would incur.
|
||||||
|
type pointerReporter struct {
|
||||||
|
last *events.DropEvent
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *pointerReporter) ReportDrop(e events.DropEvent) {
|
||||||
|
r.last = &e
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *pointerReporter) ReportFlowCreate(events.FlowCreateEvent) {}
|
||||||
|
func (r *pointerReporter) ReportFlowEvict(events.FlowEvictEvent) {}
|
||||||
|
func (r *pointerReporter) ReportRulesReload(events.RulesReloadEvent) {
|
||||||
|
}
|
||||||
|
|
||||||
|
func newBenchFixture(b *testing.B) *eventFixture {
|
||||||
|
b.Helper()
|
||||||
|
l := test.NewLogger()
|
||||||
|
|
||||||
|
vpnNetworks := new(bart.Lite)
|
||||||
|
vpnNetworks.Insert(netip.MustParsePrefix("1.2.3.0/24"))
|
||||||
|
|
||||||
|
c := &dummyCert{
|
||||||
|
name: "host1",
|
||||||
|
networks: []netip.Prefix{netip.MustParsePrefix("1.2.3.4/24")},
|
||||||
|
groups: []string{"default-group"},
|
||||||
|
issuer: "signer-shasum",
|
||||||
|
}
|
||||||
|
h := &HostInfo{
|
||||||
|
ConnectionState: &ConnectionState{
|
||||||
|
peerCert: &cert.CachedCertificate{
|
||||||
|
Certificate: c,
|
||||||
|
InvertedGroups: map[string]struct{}{"default-group": {}},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
vpnAddrs: []netip.Addr{netip.MustParseAddr("1.2.3.4")},
|
||||||
|
}
|
||||||
|
h.buildNetworks(vpnNetworks, c)
|
||||||
|
|
||||||
|
fw := NewFirewall(l, time.Minute, time.Minute, time.Minute, c)
|
||||||
|
// Inbound rule that matches our packet; outbound has no match so we can
|
||||||
|
// also benchmark the no-rule drop path.
|
||||||
|
if err := fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"any"}, "", "", "", "", ""); err != nil {
|
||||||
|
b.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
ctl := &Control{f: &Interface{firewall: fw}, l: l}
|
||||||
|
return &eventFixture{
|
||||||
|
ctl: ctl,
|
||||||
|
fw: fw,
|
||||||
|
p: firewall.Packet{
|
||||||
|
LocalAddr: netip.MustParseAddr("1.2.3.4"),
|
||||||
|
RemoteAddr: netip.MustParseAddr("1.2.3.4"),
|
||||||
|
LocalPort: 10,
|
||||||
|
RemotePort: 90,
|
||||||
|
Protocol: firewall.ProtoUDP,
|
||||||
|
},
|
||||||
|
h: h,
|
||||||
|
cp: cert.NewCAPool(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// BenchmarkFirewallDropPath measures the cost of Firewall.Drop on a packet
|
||||||
|
// that reaches the no-matching-rule branch (the longest drop path). Compare
|
||||||
|
// reporter shapes:
|
||||||
|
//
|
||||||
|
// nilReporter -- no reporter installed (feature cost when off)
|
||||||
|
// noopReporter -- reporter installed, methods discard args (minimum on-cost)
|
||||||
|
// bufferedReporter -- realistic zero-alloc reporter: value-typed channels
|
||||||
|
// pointerReporter -- anti-pattern that takes &composite-literal (allocates)
|
||||||
|
func BenchmarkFirewallDropPath(b *testing.B) {
|
||||||
|
run := func(b *testing.B, install func(*Control)) {
|
||||||
|
f := newBenchFixture(b)
|
||||||
|
install(f.ctl)
|
||||||
|
b.ReportAllocs()
|
||||||
|
b.ResetTimer()
|
||||||
|
for i := 0; i < b.N; i++ {
|
||||||
|
_ = f.firewall().Drop(f.p, firewall.PacketContext{}, false, f.h, f.cp, nil)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
b.Run("nilReporter", func(b *testing.B) { run(b, func(*Control) {}) })
|
||||||
|
b.Run("noopReporter", func(b *testing.B) {
|
||||||
|
run(b, func(c *Control) { c.SetFirewallEventReporter(noopReporter{}) })
|
||||||
|
})
|
||||||
|
b.Run("bufferedReporter", func(b *testing.B) {
|
||||||
|
run(b, func(c *Control) { c.SetFirewallEventReporter(newBufferedReporter(1024)) })
|
||||||
|
})
|
||||||
|
b.Run("pointerReporter", func(b *testing.B) {
|
||||||
|
run(b, func(c *Control) { c.SetFirewallEventReporter(&pointerReporter{}) })
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// BenchmarkConntrackCreate measures Firewall.Drop for an allowed inbound
|
||||||
|
// packet on a fresh conntrack (so addConn fires each iteration).
|
||||||
|
func BenchmarkConntrackCreate(b *testing.B) {
|
||||||
|
run := func(b *testing.B, install func(*Control)) {
|
||||||
|
f := newBenchFixture(b)
|
||||||
|
install(f.ctl)
|
||||||
|
b.ReportAllocs()
|
||||||
|
b.ResetTimer()
|
||||||
|
for i := 0; i < b.N; i++ {
|
||||||
|
resetConntrack(f.firewall())
|
||||||
|
_ = f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
b.Run("nilReporter", func(b *testing.B) { run(b, func(*Control) {}) })
|
||||||
|
b.Run("noopReporter", func(b *testing.B) {
|
||||||
|
run(b, func(c *Control) { c.SetFirewallEventReporter(noopReporter{}) })
|
||||||
|
})
|
||||||
|
b.Run("bufferedReporter", func(b *testing.B) {
|
||||||
|
run(b, func(c *Control) { c.SetFirewallEventReporter(newBufferedReporter(1024)) })
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// BenchmarkConntrackHit measures the hot path where a flow is already in
|
||||||
|
// conntrack and short-circuits rule evaluation. The reporter slot is checked
|
||||||
|
// only on create/evict, so this bench should show the reporter having zero
|
||||||
|
// impact regardless of install state.
|
||||||
|
func BenchmarkConntrackHit(b *testing.B) {
|
||||||
|
b.Run("nilReporter", func(b *testing.B) {
|
||||||
|
f := newBenchFixture(b)
|
||||||
|
// Prime conntrack.
|
||||||
|
require.NoError(b, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
b.ReportAllocs()
|
||||||
|
b.ResetTimer()
|
||||||
|
for i := 0; i < b.N; i++ {
|
||||||
|
_ = f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
b.Run("noopReporter", func(b *testing.B) {
|
||||||
|
f := newBenchFixture(b)
|
||||||
|
f.ctl.SetFirewallEventReporter(noopReporter{})
|
||||||
|
require.NoError(b, f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil))
|
||||||
|
b.ReportAllocs()
|
||||||
|
b.ResetTimer()
|
||||||
|
for i := 0; i < b.N; i++ {
|
||||||
|
_ = f.firewall().Drop(f.p, firewall.PacketContext{}, true, f.h, f.cp, nil)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
+52
-52
@@ -213,44 +213,44 @@ func TestFirewall_Drop(t *testing.T) {
|
|||||||
cp := cert.NewCAPool()
|
cp := cert.NewCAPool()
|
||||||
|
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, ErrNoMatchingRule, fw.Drop(p, false, &h, cp, nil))
|
assert.Equal(t, ErrNoMatchingRule, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
// Allow outbound because conntrack
|
// Allow outbound because conntrack
|
||||||
require.NoError(t, fw.Drop(p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
|
|
||||||
// test remote mismatch
|
// test remote mismatch
|
||||||
oldRemote := p.RemoteAddr
|
oldRemote := p.RemoteAddr
|
||||||
p.RemoteAddr = netip.MustParseAddr("1.2.3.10")
|
p.RemoteAddr = netip.MustParseAddr("1.2.3.10")
|
||||||
assert.Equal(t, fw.Drop(p, false, &h, cp, nil), ErrInvalidRemoteIP)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil), ErrInvalidRemoteIP)
|
||||||
p.RemoteAddr = oldRemote
|
p.RemoteAddr = oldRemote
|
||||||
|
|
||||||
// ensure signer doesn't get in the way of group checks
|
// ensure signer doesn't get in the way of group checks
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum"))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum-bad"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum-bad"))
|
||||||
assert.Equal(t, fw.Drop(p, true, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil), ErrNoMatchingRule)
|
||||||
|
|
||||||
// test caSha doesn't drop on match
|
// test caSha doesn't drop on match
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum-bad"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum-bad"))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum"))
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
|
|
||||||
// ensure ca name doesn't get in the way of group checks
|
// ensure ca name doesn't get in the way of group checks
|
||||||
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good", ""))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good-bad", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good-bad", ""))
|
||||||
assert.Equal(t, fw.Drop(p, true, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil), ErrNoMatchingRule)
|
||||||
|
|
||||||
// test caName doesn't drop on match
|
// test caName doesn't drop on match
|
||||||
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good-bad", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good-bad", ""))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good", ""))
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestFirewall_DropV6(t *testing.T) {
|
func TestFirewall_DropV6(t *testing.T) {
|
||||||
@@ -292,44 +292,44 @@ func TestFirewall_DropV6(t *testing.T) {
|
|||||||
cp := cert.NewCAPool()
|
cp := cert.NewCAPool()
|
||||||
|
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, ErrNoMatchingRule, fw.Drop(p, false, &h, cp, nil))
|
assert.Equal(t, ErrNoMatchingRule, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
// Allow outbound because conntrack
|
// Allow outbound because conntrack
|
||||||
require.NoError(t, fw.Drop(p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
|
|
||||||
// test remote mismatch
|
// test remote mismatch
|
||||||
oldRemote := p.RemoteAddr
|
oldRemote := p.RemoteAddr
|
||||||
p.RemoteAddr = netip.MustParseAddr("fd12::56")
|
p.RemoteAddr = netip.MustParseAddr("fd12::56")
|
||||||
assert.Equal(t, fw.Drop(p, false, &h, cp, nil), ErrInvalidRemoteIP)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil), ErrInvalidRemoteIP)
|
||||||
p.RemoteAddr = oldRemote
|
p.RemoteAddr = oldRemote
|
||||||
|
|
||||||
// ensure signer doesn't get in the way of group checks
|
// ensure signer doesn't get in the way of group checks
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum"))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum-bad"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum-bad"))
|
||||||
assert.Equal(t, fw.Drop(p, true, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil), ErrNoMatchingRule)
|
||||||
|
|
||||||
// test caSha doesn't drop on match
|
// test caSha doesn't drop on match
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum-bad"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "", "signer-shasum-bad"))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum"))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "", "signer-shasum"))
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
|
|
||||||
// ensure ca name doesn't get in the way of group checks
|
// ensure ca name doesn't get in the way of group checks
|
||||||
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good", ""))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good-bad", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good-bad", ""))
|
||||||
assert.Equal(t, fw.Drop(p, true, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil), ErrNoMatchingRule)
|
||||||
|
|
||||||
// test caName doesn't drop on match
|
// test caName doesn't drop on match
|
||||||
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
cp.CAs["signer-shasum"] = &cert.CachedCertificate{Certificate: &dummyCert{name: "ca-good"}}
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, &c)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good-bad", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"nope"}, "", "", "", "ca-good-bad", ""))
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 0, 0, []string{"default-group"}, "", "", "", "ca-good", ""))
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
}
|
}
|
||||||
|
|
||||||
func BenchmarkFirewallTable_match(b *testing.B) {
|
func BenchmarkFirewallTable_match(b *testing.B) {
|
||||||
@@ -537,10 +537,10 @@ func TestFirewall_Drop2(t *testing.T) {
|
|||||||
cp := cert.NewCAPool()
|
cp := cert.NewCAPool()
|
||||||
|
|
||||||
// h1/c1 lacks the proper groups
|
// h1/c1 lacks the proper groups
|
||||||
require.ErrorIs(t, fw.Drop(p, true, &h1, cp, nil), ErrNoMatchingRule)
|
require.ErrorIs(t, fw.Drop(p, firewall.PacketContext{}, true, &h1, cp, nil), ErrNoMatchingRule)
|
||||||
// c has the proper groups
|
// c has the proper groups
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestFirewall_Drop3(t *testing.T) {
|
func TestFirewall_Drop3(t *testing.T) {
|
||||||
@@ -618,18 +618,18 @@ func TestFirewall_Drop3(t *testing.T) {
|
|||||||
cp := cert.NewCAPool()
|
cp := cert.NewCAPool()
|
||||||
|
|
||||||
// c1 should pass because host match
|
// c1 should pass because host match
|
||||||
require.NoError(t, fw.Drop(p, true, &h1, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h1, cp, nil))
|
||||||
// c2 should pass because ca sha match
|
// c2 should pass because ca sha match
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(p, true, &h2, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h2, cp, nil))
|
||||||
// c3 should fail because no match
|
// c3 should fail because no match
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
assert.Equal(t, fw.Drop(p, true, &h3, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, true, &h3, cp, nil), ErrNoMatchingRule)
|
||||||
|
|
||||||
// Test a remote address match
|
// Test a remote address match
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 1, 1, []string{}, "", "1.2.3.4/24", "", "", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 1, 1, []string{}, "", "1.2.3.4/24", "", "", ""))
|
||||||
require.NoError(t, fw.Drop(p, true, &h1, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h1, cp, nil))
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestFirewall_Drop3V6(t *testing.T) {
|
func TestFirewall_Drop3V6(t *testing.T) {
|
||||||
@@ -667,7 +667,7 @@ func TestFirewall_Drop3V6(t *testing.T) {
|
|||||||
fw := NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
fw := NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
||||||
cp := cert.NewCAPool()
|
cp := cert.NewCAPool()
|
||||||
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 1, 1, []string{}, "", "fd12::34/120", "", "", ""))
|
require.NoError(t, fw.AddRule(true, firewall.ProtoAny, 1, 1, []string{}, "", "fd12::34/120", "", "", ""))
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestFirewall_DropConntrackReload(t *testing.T) {
|
func TestFirewall_DropConntrackReload(t *testing.T) {
|
||||||
@@ -709,12 +709,12 @@ func TestFirewall_DropConntrackReload(t *testing.T) {
|
|||||||
cp := cert.NewCAPool()
|
cp := cert.NewCAPool()
|
||||||
|
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
// Allow outbound because conntrack
|
// Allow outbound because conntrack
|
||||||
require.NoError(t, fw.Drop(p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
|
|
||||||
oldFw := fw
|
oldFw := fw
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
||||||
@@ -723,7 +723,7 @@ func TestFirewall_DropConntrackReload(t *testing.T) {
|
|||||||
fw.rulesVersion = oldFw.rulesVersion + 1
|
fw.rulesVersion = oldFw.rulesVersion + 1
|
||||||
|
|
||||||
// Allow outbound because conntrack and new rules allow port 10
|
// Allow outbound because conntrack and new rules allow port 10
|
||||||
require.NoError(t, fw.Drop(p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
|
|
||||||
oldFw = fw
|
oldFw = fw
|
||||||
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
fw = NewFirewall(l, time.Second, time.Minute, time.Hour, c.Certificate)
|
||||||
@@ -732,7 +732,7 @@ func TestFirewall_DropConntrackReload(t *testing.T) {
|
|||||||
fw.rulesVersion = oldFw.rulesVersion + 1
|
fw.rulesVersion = oldFw.rulesVersion + 1
|
||||||
|
|
||||||
// Drop outbound because conntrack doesn't match new ruleset
|
// Drop outbound because conntrack doesn't match new ruleset
|
||||||
assert.Equal(t, fw.Drop(p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
||||||
@@ -778,12 +778,12 @@ func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
|||||||
p.LocalPort = 0
|
p.LocalPort = 0
|
||||||
p.RemotePort = 0
|
p.RemotePort = 0
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(*p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
//now also allow outbound
|
//now also allow outbound
|
||||||
require.NoError(t, fw.Drop(*p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
})
|
})
|
||||||
|
|
||||||
t.Run("nonzero ports", func(t *testing.T) {
|
t.Run("nonzero ports", func(t *testing.T) {
|
||||||
@@ -791,12 +791,12 @@ func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
|||||||
p.LocalPort = 0xabcd
|
p.LocalPort = 0xabcd
|
||||||
p.RemotePort = 0x1234
|
p.RemotePort = 0x1234
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(*p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
//now also allow outbound
|
//now also allow outbound
|
||||||
require.NoError(t, fw.Drop(*p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
@@ -808,12 +808,12 @@ func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
|||||||
p.LocalPort = 0
|
p.LocalPort = 0
|
||||||
p.RemotePort = 0
|
p.RemotePort = 0
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
assert.Equal(t, fw.Drop(*p, true, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, true, &h, cp, nil), ErrNoMatchingRule)
|
||||||
//now also allow outbound
|
//now also allow outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
})
|
})
|
||||||
|
|
||||||
t.Run("nonzero ports, still blocked", func(t *testing.T) {
|
t.Run("nonzero ports, still blocked", func(t *testing.T) {
|
||||||
@@ -821,12 +821,12 @@ func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
|||||||
p.LocalPort = 0xabcd
|
p.LocalPort = 0xabcd
|
||||||
p.RemotePort = 0x1234
|
p.RemotePort = 0x1234
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
assert.Equal(t, fw.Drop(*p, true, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, true, &h, cp, nil), ErrNoMatchingRule)
|
||||||
//now also allow outbound
|
//now also allow outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
})
|
})
|
||||||
|
|
||||||
t.Run("nonzero, matching ports, still blocked", func(t *testing.T) {
|
t.Run("nonzero, matching ports, still blocked", func(t *testing.T) {
|
||||||
@@ -834,12 +834,12 @@ func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
|||||||
p.LocalPort = 80
|
p.LocalPort = 80
|
||||||
p.RemotePort = 80
|
p.RemotePort = 80
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
assert.Equal(t, fw.Drop(*p, true, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, true, &h, cp, nil), ErrNoMatchingRule)
|
||||||
//now also allow outbound
|
//now also allow outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
t.Run("Any proto, any port", func(t *testing.T) {
|
t.Run("Any proto, any port", func(t *testing.T) {
|
||||||
@@ -851,12 +851,12 @@ func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
|||||||
p.LocalPort = 0
|
p.LocalPort = 0
|
||||||
p.RemotePort = 0
|
p.RemotePort = 0
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(*p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
//now also allow outbound
|
//now also allow outbound
|
||||||
require.NoError(t, fw.Drop(*p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
})
|
})
|
||||||
|
|
||||||
t.Run("nonzero ports, allowed", func(t *testing.T) {
|
t.Run("nonzero ports, allowed", func(t *testing.T) {
|
||||||
@@ -865,15 +865,15 @@ func TestFirewall_ICMPPortBehavior(t *testing.T) {
|
|||||||
p.LocalPort = 0xabcd
|
p.LocalPort = 0xabcd
|
||||||
p.RemotePort = 0x1234
|
p.RemotePort = 0x1234
|
||||||
// Drop outbound
|
// Drop outbound
|
||||||
assert.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
assert.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
// Allow inbound
|
// Allow inbound
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
require.NoError(t, fw.Drop(*p, true, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, true, &h, cp, nil))
|
||||||
//now also allow outbound
|
//now also allow outbound
|
||||||
require.NoError(t, fw.Drop(*p, false, &h, cp, nil))
|
require.NoError(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil))
|
||||||
//different ID is blocked
|
//different ID is blocked
|
||||||
p.RemotePort++
|
p.RemotePort++
|
||||||
require.Equal(t, fw.Drop(*p, false, &h, cp, nil), ErrNoMatchingRule)
|
require.Equal(t, fw.Drop(*p, firewall.PacketContext{}, false, &h, cp, nil), ErrNoMatchingRule)
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
@@ -922,7 +922,7 @@ func TestFirewall_DropIPSpoofing(t *testing.T) {
|
|||||||
Protocol: firewall.ProtoUDP,
|
Protocol: firewall.ProtoUDP,
|
||||||
Fragment: false,
|
Fragment: false,
|
||||||
}
|
}
|
||||||
assert.Equal(t, fw.Drop(p, true, &h1, cp, nil), ErrInvalidRemoteIP)
|
assert.Equal(t, fw.Drop(p, firewall.PacketContext{}, true, &h1, cp, nil), ErrInvalidRemoteIP)
|
||||||
}
|
}
|
||||||
|
|
||||||
func BenchmarkLookup(b *testing.B) {
|
func BenchmarkLookup(b *testing.B) {
|
||||||
@@ -1336,7 +1336,7 @@ func (c *testcase) Test(t *testing.T, fw *Firewall) {
|
|||||||
t.Helper()
|
t.Helper()
|
||||||
cp := cert.NewCAPool()
|
cp := cert.NewCAPool()
|
||||||
resetConntrack(fw)
|
resetConntrack(fw)
|
||||||
err := fw.Drop(c.p, true, c.h, cp, nil)
|
err := fw.Drop(c.p, firewall.PacketContext{}, true, c.h, cp, nil)
|
||||||
if c.err == nil {
|
if c.err == nil {
|
||||||
require.NoError(t, err, "failed to not drop remote address %s", c.p.RemoteAddr)
|
require.NoError(t, err, "failed to not drop remote address %s", c.p.RemoteAddr)
|
||||||
} else {
|
} else {
|
||||||
|
|||||||
+2
-2
@@ -604,9 +604,9 @@ func (hm *HostMap) queryVpnAddr(vpnIp netip.Addr, promoteIfce *Interface) *HostI
|
|||||||
// unlockedAddHostInfo assumes you have a write-lock and will add a hostinfo object to the hostmap Indexes and RemoteIndexes maps.
|
// unlockedAddHostInfo assumes you have a write-lock and will add a hostinfo object to the hostmap Indexes and RemoteIndexes maps.
|
||||||
// If an entry exists for the Hosts table (vpnIp -> hostinfo) then the provided hostinfo will be made primary
|
// If an entry exists for the Hosts table (vpnIp -> hostinfo) then the provided hostinfo will be made primary
|
||||||
func (hm *HostMap) unlockedAddHostInfo(hostinfo *HostInfo, f *Interface) {
|
func (hm *HostMap) unlockedAddHostInfo(hostinfo *HostInfo, f *Interface) {
|
||||||
if f.serveDns {
|
if f.dnsServer != nil {
|
||||||
remoteCert := hostinfo.ConnectionState.peerCert
|
remoteCert := hostinfo.ConnectionState.peerCert
|
||||||
dnsR.Add(remoteCert.Certificate.Name()+".", hostinfo.vpnAddrs)
|
f.dnsServer.Add(remoteCert.Certificate.Name()+".", hostinfo.vpnAddrs)
|
||||||
}
|
}
|
||||||
for _, addr := range hostinfo.vpnAddrs {
|
for _, addr := range hostinfo.vpnAddrs {
|
||||||
hm.unlockedInnerAddHostInfo(addr, hostinfo, f)
|
hm.unlockedInnerAddHostInfo(addr, hostinfo, f)
|
||||||
|
|||||||
@@ -11,8 +11,8 @@ import (
|
|||||||
"github.com/slackhq/nebula/routing"
|
"github.com/slackhq/nebula/routing"
|
||||||
)
|
)
|
||||||
|
|
||||||
func (f *Interface) consumeInsidePacket(packet []byte, fwPacket *firewall.Packet, nb []byte, batch *sendBatch, rejectBuf []byte, q int, localCache firewall.ConntrackCache) {
|
func (f *Interface) consumeInsidePacket(packet []byte, fwPacket *firewall.Packet, fwCtx *firewall.PacketContext, nb, out []byte, q int, localCache firewall.ConntrackCache) {
|
||||||
err := newPacket(packet, false, fwPacket)
|
err := newPacket(packet, false, fwPacket, fwCtx)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if f.l.Level >= logrus.DebugLevel {
|
if f.l.Level >= logrus.DebugLevel {
|
||||||
f.l.WithField("packet", packet).Debugf("Error while validating outbound packet: %s", err)
|
f.l.WithField("packet", packet).Debugf("Error while validating outbound packet: %s", err)
|
||||||
@@ -33,7 +33,7 @@ func (f *Interface) consumeInsidePacket(packet []byte, fwPacket *firewall.Packet
|
|||||||
// routes packets from the Nebula addr to the Nebula addr through the Nebula
|
// routes packets from the Nebula addr to the Nebula addr through the Nebula
|
||||||
// TUN device.
|
// TUN device.
|
||||||
if immediatelyForwardToSelf {
|
if immediatelyForwardToSelf {
|
||||||
_, err := f.readers[q].WriteReject(packet)
|
_, err := f.readers[q].Write(packet)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
f.l.WithError(err).Error("Failed to forward to tun")
|
f.l.WithError(err).Error("Failed to forward to tun")
|
||||||
}
|
}
|
||||||
@@ -53,7 +53,7 @@ func (f *Interface) consumeInsidePacket(packet []byte, fwPacket *firewall.Packet
|
|||||||
})
|
})
|
||||||
|
|
||||||
if hostinfo == nil {
|
if hostinfo == nil {
|
||||||
f.rejectInside(packet, rejectBuf, q)
|
f.rejectInside(packet, out, q)
|
||||||
if f.l.Level >= logrus.DebugLevel {
|
if f.l.Level >= logrus.DebugLevel {
|
||||||
f.l.WithField("vpnAddr", fwPacket.RemoteAddr).
|
f.l.WithField("vpnAddr", fwPacket.RemoteAddr).
|
||||||
WithField("fwPacket", fwPacket).
|
WithField("fwPacket", fwPacket).
|
||||||
@@ -66,12 +66,12 @@ func (f *Interface) consumeInsidePacket(packet []byte, fwPacket *firewall.Packet
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
dropReason := f.firewall.Drop(*fwPacket, false, hostinfo, f.pki.GetCAPool(), localCache)
|
dropReason := f.firewall.Drop(*fwPacket, *fwCtx, false, hostinfo, f.pki.GetCAPool(), localCache)
|
||||||
if dropReason == nil {
|
if dropReason == nil {
|
||||||
f.sendInsideMessage(hostinfo, packet, nb, batch, rejectBuf, q)
|
f.sendNoMetrics(header.Message, 0, hostinfo.ConnectionState, hostinfo, netip.AddrPort{}, packet, nb, out, q)
|
||||||
|
|
||||||
} else {
|
} else {
|
||||||
f.rejectInside(packet, rejectBuf, q)
|
f.rejectInside(packet, out, q)
|
||||||
if f.l.Level >= logrus.DebugLevel {
|
if f.l.Level >= logrus.DebugLevel {
|
||||||
hostinfo.logger(f.l).
|
hostinfo.logger(f.l).
|
||||||
WithField("fwPacket", fwPacket).
|
WithField("fwPacket", fwPacket).
|
||||||
@@ -81,63 +81,6 @@ func (f *Interface) consumeInsidePacket(packet []byte, fwPacket *firewall.Packet
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// sendInsideMessage encrypts a firewall-approved inside packet into the
|
|
||||||
// caller's batch slot for later sendmmsg flush. When hostinfo.remote is not
|
|
||||||
// valid we fall through to the relay slow path via the unbatched sendNoMetrics
|
|
||||||
// so relay behavior is unchanged.
|
|
||||||
func (f *Interface) sendInsideMessage(hostinfo *HostInfo, p, nb []byte, batch *sendBatch, rejectBuf []byte, q int) {
|
|
||||||
ci := hostinfo.ConnectionState
|
|
||||||
if ci.eKey == nil {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
if !hostinfo.remote.IsValid() {
|
|
||||||
// Slow path: relay fallback. Reuse rejectBuf as the ciphertext
|
|
||||||
// scratch; sendNoMetrics arranges header space for SendVia.
|
|
||||||
f.sendNoMetrics(header.Message, 0, ci, hostinfo, netip.AddrPort{}, p, nb, rejectBuf, q)
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
scratch := batch.Next()
|
|
||||||
if scratch == nil {
|
|
||||||
// Batch full: bypass batching and send this packet directly so we
|
|
||||||
// never drop traffic on over-subscribed iterations.
|
|
||||||
f.sendNoMetrics(header.Message, 0, ci, hostinfo, netip.AddrPort{}, p, nb, rejectBuf, q)
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
if noiseutil.EncryptLockNeeded {
|
|
||||||
ci.writeLock.Lock()
|
|
||||||
}
|
|
||||||
c := ci.messageCounter.Add(1)
|
|
||||||
|
|
||||||
out := header.Encode(scratch, header.Version, header.Message, 0, hostinfo.remoteIndexId, c)
|
|
||||||
f.connectionManager.Out(hostinfo)
|
|
||||||
|
|
||||||
if hostinfo.lastRebindCount != f.rebindCount {
|
|
||||||
//NOTE: there is an update hole if a tunnel isn't used and exactly 256 rebinds occur before the tunnel is
|
|
||||||
// finally used again. This tunnel would eventually be torn down and recreated if this action didn't help.
|
|
||||||
f.lightHouse.QueryServer(hostinfo.vpnAddrs[0])
|
|
||||||
hostinfo.lastRebindCount = f.rebindCount
|
|
||||||
if f.l.Level >= logrus.DebugLevel {
|
|
||||||
f.l.WithField("vpnAddrs", hostinfo.vpnAddrs).Debug("Lighthouse update triggered for punch due to rebind counter")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
out, err := ci.eKey.EncryptDanger(out, out, p, c, nb)
|
|
||||||
if noiseutil.EncryptLockNeeded {
|
|
||||||
ci.writeLock.Unlock()
|
|
||||||
}
|
|
||||||
if err != nil {
|
|
||||||
hostinfo.logger(f.l).WithError(err).
|
|
||||||
WithField("udpAddr", hostinfo.remote).WithField("counter", c).
|
|
||||||
Error("Failed to encrypt outgoing packet")
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
batch.Commit(len(out), hostinfo.remote)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (f *Interface) rejectInside(packet []byte, out []byte, q int) {
|
func (f *Interface) rejectInside(packet []byte, out []byte, q int) {
|
||||||
if !f.firewall.InSendReject {
|
if !f.firewall.InSendReject {
|
||||||
return
|
return
|
||||||
@@ -148,7 +91,7 @@ func (f *Interface) rejectInside(packet []byte, out []byte, q int) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
_, err := f.readers[q].WriteReject(out)
|
_, err := f.readers[q].Write(out)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
f.l.WithError(err).Error("Failed to write to tun")
|
f.l.WithError(err).Error("Failed to write to tun")
|
||||||
}
|
}
|
||||||
@@ -268,14 +211,15 @@ func (f *Interface) getOrHandshakeConsiderRouting(fwPacket *firewall.Packet, cac
|
|||||||
|
|
||||||
func (f *Interface) sendMessageNow(t header.MessageType, st header.MessageSubType, hostinfo *HostInfo, p, nb, out []byte) {
|
func (f *Interface) sendMessageNow(t header.MessageType, st header.MessageSubType, hostinfo *HostInfo, p, nb, out []byte) {
|
||||||
fp := &firewall.Packet{}
|
fp := &firewall.Packet{}
|
||||||
err := newPacket(p, false, fp)
|
ctx := &firewall.PacketContext{}
|
||||||
|
err := newPacket(p, false, fp, ctx)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
f.l.Warnf("error while parsing outgoing packet for firewall check; %v", err)
|
f.l.Warnf("error while parsing outgoing packet for firewall check; %v", err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
// check if packet is in outbound fw rules
|
// check if packet is in outbound fw rules
|
||||||
dropReason := f.firewall.Drop(*fp, false, hostinfo, f.pki.GetCAPool(), nil)
|
dropReason := f.firewall.Drop(*fp, *ctx, false, hostinfo, f.pki.GetCAPool(), nil)
|
||||||
if dropReason != nil {
|
if dropReason != nil {
|
||||||
if f.l.Level >= logrus.DebugLevel {
|
if f.l.Level >= logrus.DebugLevel {
|
||||||
f.l.WithField("fwPacket", fp).
|
f.l.WithField("fwPacket", fp).
|
||||||
|
|||||||
+28
-96
@@ -4,6 +4,7 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"sync"
|
"sync"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
@@ -28,7 +29,7 @@ type InterfaceConfig struct {
|
|||||||
pki *PKI
|
pki *PKI
|
||||||
Cipher string
|
Cipher string
|
||||||
Firewall *Firewall
|
Firewall *Firewall
|
||||||
ServeDns bool
|
DnsServer *dnsServer
|
||||||
HandshakeManager *HandshakeManager
|
HandshakeManager *HandshakeManager
|
||||||
lightHouse *LightHouse
|
lightHouse *LightHouse
|
||||||
connectionManager *connectionManager
|
connectionManager *connectionManager
|
||||||
@@ -56,7 +57,7 @@ type Interface struct {
|
|||||||
firewall *Firewall
|
firewall *Firewall
|
||||||
connectionManager *connectionManager
|
connectionManager *connectionManager
|
||||||
handshakeManager *HandshakeManager
|
handshakeManager *HandshakeManager
|
||||||
serveDns bool
|
dnsServer *dnsServer
|
||||||
createTime time.Time
|
createTime time.Time
|
||||||
lightHouse *LightHouse
|
lightHouse *LightHouse
|
||||||
myBroadcastAddrsTable *bart.Lite
|
myBroadcastAddrsTable *bart.Lite
|
||||||
@@ -84,13 +85,10 @@ type Interface struct {
|
|||||||
|
|
||||||
conntrackCacheTimeout time.Duration
|
conntrackCacheTimeout time.Duration
|
||||||
|
|
||||||
|
ctx context.Context
|
||||||
writers []udp.Conn
|
writers []udp.Conn
|
||||||
readers []overlay.Queue
|
readers []io.ReadWriteCloser
|
||||||
// tunCoalescers is one tcpCoalescer per tun queue, wrapping readers[i].
|
wg sync.WaitGroup
|
||||||
// decryptToTun sends plaintext into the coalescer; listenOut calls its
|
|
||||||
// Flush at the end of each UDP recvmmsg batch.
|
|
||||||
tunCoalescers []*tcpCoalescer
|
|
||||||
wg sync.WaitGroup
|
|
||||||
|
|
||||||
// fatalErr holds the first unexpected reader error that caused shutdown.
|
// fatalErr holds the first unexpected reader error that caused shutdown.
|
||||||
// nil means "no fatal error" (yet)
|
// nil means "no fatal error" (yet)
|
||||||
@@ -173,12 +171,13 @@ func NewInterface(ctx context.Context, c *InterfaceConfig) (*Interface, error) {
|
|||||||
|
|
||||||
cs := c.pki.getCertState()
|
cs := c.pki.getCertState()
|
||||||
ifce := &Interface{
|
ifce := &Interface{
|
||||||
|
ctx: ctx,
|
||||||
pki: c.pki,
|
pki: c.pki,
|
||||||
hostMap: c.HostMap,
|
hostMap: c.HostMap,
|
||||||
outside: c.Outside,
|
outside: c.Outside,
|
||||||
inside: c.Inside,
|
inside: c.Inside,
|
||||||
firewall: c.Firewall,
|
firewall: c.Firewall,
|
||||||
serveDns: c.ServeDns,
|
dnsServer: c.DnsServer,
|
||||||
handshakeManager: c.HandshakeManager,
|
handshakeManager: c.HandshakeManager,
|
||||||
createTime: time.Now(),
|
createTime: time.Now(),
|
||||||
lightHouse: c.lightHouse,
|
lightHouse: c.lightHouse,
|
||||||
@@ -187,8 +186,7 @@ func NewInterface(ctx context.Context, c *InterfaceConfig) (*Interface, error) {
|
|||||||
routines: c.routines,
|
routines: c.routines,
|
||||||
version: c.version,
|
version: c.version,
|
||||||
writers: make([]udp.Conn, c.routines),
|
writers: make([]udp.Conn, c.routines),
|
||||||
readers: make([]overlay.Queue, c.routines),
|
readers: make([]io.ReadWriteCloser, c.routines),
|
||||||
tunCoalescers: make([]*tcpCoalescer, c.routines),
|
|
||||||
myVpnNetworks: cs.myVpnNetworks,
|
myVpnNetworks: cs.myVpnNetworks,
|
||||||
myVpnNetworksTable: cs.myVpnNetworksTable,
|
myVpnNetworksTable: cs.myVpnNetworksTable,
|
||||||
myVpnAddrs: cs.myVpnAddrs,
|
myVpnAddrs: cs.myVpnAddrs,
|
||||||
@@ -243,7 +241,7 @@ func (f *Interface) activate() error {
|
|||||||
metrics.GetOrRegisterGauge("routines", nil).Update(int64(f.routines))
|
metrics.GetOrRegisterGauge("routines", nil).Update(int64(f.routines))
|
||||||
|
|
||||||
// Prepare n tun queues
|
// Prepare n tun queues
|
||||||
var reader overlay.Queue = f.inside
|
var reader io.ReadWriteCloser = f.inside
|
||||||
for i := 0; i < f.routines; i++ {
|
for i := 0; i < f.routines; i++ {
|
||||||
if i > 0 {
|
if i > 0 {
|
||||||
reader, err = f.inside.NewMultiQueueReader()
|
reader, err = f.inside.NewMultiQueueReader()
|
||||||
@@ -252,7 +250,6 @@ func (f *Interface) activate() error {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
f.readers[i] = reader
|
f.readers[i] = reader
|
||||||
f.tunCoalescers[i] = newTCPCoalescer(reader)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
f.wg.Add(1) // for us to wait on Close() to return
|
f.wg.Add(1) // for us to wait on Close() to return
|
||||||
@@ -308,30 +305,16 @@ func (f *Interface) listenOut(i int) {
|
|||||||
li = f.outside
|
li = f.outside
|
||||||
}
|
}
|
||||||
|
|
||||||
ctCache := firewall.NewConntrackCacheTicker(f.conntrackCacheTimeout)
|
ctCache := firewall.NewConntrackCacheTicker(f.ctx, f.conntrackCacheTimeout)
|
||||||
lhh := f.lightHouse.NewRequestHandler()
|
lhh := f.lightHouse.NewRequestHandler()
|
||||||
|
plaintext := make([]byte, udp.MTU)
|
||||||
h := &header.H{}
|
h := &header.H{}
|
||||||
fwPacket := &firewall.Packet{}
|
fwPacket := &firewall.Packet{}
|
||||||
|
fwCtx := &firewall.PacketContext{}
|
||||||
nb := make([]byte, 12, 12)
|
nb := make([]byte, 12, 12)
|
||||||
|
|
||||||
// plaintexts is a ring of decrypt scratches, one per packet in a UDP
|
|
||||||
// recvmmsg batch. The coalescer borrows payload slices from here and
|
|
||||||
// requires they stay valid until Flush — so we rotate each packet and
|
|
||||||
// reset only in the batch-end flush callback.
|
|
||||||
var plaintexts [][]byte
|
|
||||||
idx := 0
|
|
||||||
coalescer := f.tunCoalescers[i]
|
|
||||||
err := li.ListenOut(func(fromUdpAddr netip.AddrPort, payload []byte) {
|
err := li.ListenOut(func(fromUdpAddr netip.AddrPort, payload []byte) {
|
||||||
if idx >= len(plaintexts) {
|
f.readOutsidePackets(ViaSender{UdpAddr: fromUdpAddr}, plaintext[:0], payload, h, fwPacket, fwCtx, lhh, nb, i, ctCache.Get(f.l))
|
||||||
plaintexts = append(plaintexts, make([]byte, udp.MTU))
|
|
||||||
}
|
|
||||||
f.readOutsidePackets(ViaSender{UdpAddr: fromUdpAddr}, plaintexts[idx][:0], payload, h, fwPacket, lhh, nb, i, ctCache.Get(f.l))
|
|
||||||
idx++
|
|
||||||
}, func() {
|
|
||||||
if err := coalescer.Flush(); err != nil {
|
|
||||||
f.l.WithError(err).Error("Failed to flush tun coalescer")
|
|
||||||
}
|
|
||||||
idx = 0
|
|
||||||
})
|
})
|
||||||
|
|
||||||
if err != nil && !f.closed.Load() {
|
if err != nil && !f.closed.Load() {
|
||||||
@@ -342,16 +325,17 @@ func (f *Interface) listenOut(i int) {
|
|||||||
f.l.Debugf("underlay reader %v is done", i)
|
f.l.Debugf("underlay reader %v is done", i)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *Interface) listenIn(reader overlay.Queue, i int) {
|
func (f *Interface) listenIn(reader io.ReadWriteCloser, i int) {
|
||||||
rejectBuf := make([]byte, mtu)
|
packet := make([]byte, mtu)
|
||||||
batch := newSendBatch(sendBatchCap, udp.MTU+32)
|
out := make([]byte, mtu)
|
||||||
fwPacket := &firewall.Packet{}
|
fwPacket := &firewall.Packet{}
|
||||||
|
fwCtx := &firewall.PacketContext{}
|
||||||
nb := make([]byte, 12, 12)
|
nb := make([]byte, 12, 12)
|
||||||
|
|
||||||
conntrackCache := firewall.NewConntrackCacheTicker(f.conntrackCacheTimeout)
|
conntrackCache := firewall.NewConntrackCacheTicker(f.ctx, f.conntrackCacheTimeout)
|
||||||
|
|
||||||
for {
|
for {
|
||||||
pkts, err := reader.Read()
|
n, err := reader.Read(packet)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if !f.closed.Load() {
|
if !f.closed.Load() {
|
||||||
f.l.WithError(err).WithField("reader", i).Error("Error while reading outbound packet, closing")
|
f.l.WithError(err).WithField("reader", i).Error("Error while reading outbound packet, closing")
|
||||||
@@ -360,71 +344,12 @@ func (f *Interface) listenIn(reader overlay.Queue, i int) {
|
|||||||
break
|
break
|
||||||
}
|
}
|
||||||
|
|
||||||
batch.Reset()
|
f.consumeInsidePacket(packet[:n], fwPacket, fwCtx, nb, out, i, conntrackCache.Get(f.l))
|
||||||
for _, pkt := range pkts {
|
|
||||||
if batch.Len() >= batch.Cap() {
|
|
||||||
f.flushBatch(batch, i)
|
|
||||||
batch.Reset()
|
|
||||||
}
|
|
||||||
f.consumeInsidePacket(pkt, fwPacket, nb, batch, rejectBuf, i, conntrackCache.Get(f.l))
|
|
||||||
}
|
|
||||||
if batch.Len() > 0 {
|
|
||||||
f.flushBatch(batch, i)
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
f.l.Debugf("overlay reader %v is done", i)
|
f.l.Debugf("overlay reader %v is done", i)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *Interface) flushBatch(batch *sendBatch, q int) {
|
|
||||||
//if len(batch.bufs) == 1 {
|
|
||||||
// if err := f.writers[q].WriteTo(batch.bufs[0], batch.dsts[0]); err != nil {
|
|
||||||
// f.l.WithError(err).WithField("writer", q).Error("Failed to write outgoing single-batch")
|
|
||||||
// }
|
|
||||||
// return
|
|
||||||
//}
|
|
||||||
w := f.writers[q]
|
|
||||||
if w.SupportsGSO() {
|
|
||||||
if segSize, ok := batchSegmentable(batch); ok {
|
|
||||||
if err := w.WriteSegmented(batch.bufs, batch.dsts[0], segSize); err != nil {
|
|
||||||
f.l.WithError(err).WithField("writer", q).Error("Failed to write outgoing GSO batch")
|
|
||||||
}
|
|
||||||
return
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if err := w.WriteBatch(batch.bufs, batch.dsts); err != nil {
|
|
||||||
f.l.WithError(err).WithField("writer", q).Error("Failed to write outgoing batch")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// batchSegmentable reports whether a batch can be emitted as a single UDP GSO
|
|
||||||
// superpacket: all packets go to the same destination, and every packet
|
|
||||||
// except possibly the last has the same length. Returns the segment size on
|
|
||||||
// success. The single-packet case is handled in flushBatch before this runs.
|
|
||||||
func batchSegmentable(b *sendBatch) (int, bool) {
|
|
||||||
segSize := len(b.bufs[0])
|
|
||||||
if segSize == 0 {
|
|
||||||
return 0, false
|
|
||||||
}
|
|
||||||
dst := b.dsts[0]
|
|
||||||
last := len(b.bufs) - 1
|
|
||||||
for i := 1; i <= last; i++ {
|
|
||||||
if b.dsts[i] != dst {
|
|
||||||
return 0, false
|
|
||||||
}
|
|
||||||
if i < last {
|
|
||||||
if len(b.bufs[i]) != segSize {
|
|
||||||
return 0, false
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
if len(b.bufs[i]) == 0 || len(b.bufs[i]) > segSize {
|
|
||||||
return 0, false
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return segSize, true
|
|
||||||
}
|
|
||||||
|
|
||||||
func (f *Interface) RegisterConfigChangeCallbacks(c *config.C) {
|
func (f *Interface) RegisterConfigChangeCallbacks(c *config.C) {
|
||||||
c.RegisterReloadCallback(f.reloadFirewall)
|
c.RegisterReloadCallback(f.reloadFirewall)
|
||||||
c.RegisterReloadCallback(f.reloadSendRecvError)
|
c.RegisterReloadCallback(f.reloadSendRecvError)
|
||||||
@@ -477,8 +402,15 @@ func (f *Interface) reloadFirewall(c *config.C) {
|
|||||||
fw.Conntrack = conntrack
|
fw.Conntrack = conntrack
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fw.reporter = oldFw.reporter
|
||||||
|
|
||||||
f.firewall = fw
|
f.firewall = fw
|
||||||
|
|
||||||
|
// Fire ReportRulesReload under the conntrack lock so the reporter cannot
|
||||||
|
// observe a FlowCreate/FlowEvict for the new rulesVersion before it
|
||||||
|
// observes the reload marker. Report* must be non-blocking.
|
||||||
|
fw.reportRulesReload(oldFw.rulesVersion, fw.rulesVersion)
|
||||||
|
|
||||||
oldFw.Destroy()
|
oldFw.Destroy()
|
||||||
f.l.WithField("firewallHashes", fw.GetRuleHashes()).
|
f.l.WithField("firewallHashes", fw.GetRuleHashes()).
|
||||||
WithField("oldFirewallHashes", oldFw.GetRuleHashes()).
|
WithField("oldFirewallHashes", oldFw.GetRuleHashes()).
|
||||||
|
|||||||
@@ -215,13 +215,9 @@ func Main(c *config.C, configTest bool, buildVersion string, logger *logrus.Logg
|
|||||||
handshakeManager := NewHandshakeManager(l, hostMap, lightHouse, udpConns[0], handshakeConfig)
|
handshakeManager := NewHandshakeManager(l, hostMap, lightHouse, udpConns[0], handshakeConfig)
|
||||||
lightHouse.handshakeTrigger = handshakeManager.trigger
|
lightHouse.handshakeTrigger = handshakeManager.trigger
|
||||||
|
|
||||||
serveDns := false
|
ds, err := newDnsServerFromConfig(ctx, l, pki.getCertState(), hostMap, c)
|
||||||
if c.GetBool("lighthouse.serve_dns", false) {
|
if err != nil {
|
||||||
if c.GetBool("lighthouse.am_lighthouse", false) {
|
l.WithError(err).Warn("Failed to start DNS responder")
|
||||||
serveDns = true
|
|
||||||
} else {
|
|
||||||
l.Warn("DNS server refusing to run because this host is not a lighthouse.")
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
ifConfig := &InterfaceConfig{
|
ifConfig := &InterfaceConfig{
|
||||||
@@ -230,7 +226,7 @@ func Main(c *config.C, configTest bool, buildVersion string, logger *logrus.Logg
|
|||||||
Outside: udpConns[0],
|
Outside: udpConns[0],
|
||||||
pki: pki,
|
pki: pki,
|
||||||
Firewall: fw,
|
Firewall: fw,
|
||||||
ServeDns: serveDns,
|
DnsServer: ds,
|
||||||
HandshakeManager: handshakeManager,
|
HandshakeManager: handshakeManager,
|
||||||
connectionManager: connManager,
|
connectionManager: connManager,
|
||||||
lightHouse: lightHouse,
|
lightHouse: lightHouse,
|
||||||
@@ -280,13 +276,6 @@ func Main(c *config.C, configTest bool, buildVersion string, logger *logrus.Logg
|
|||||||
|
|
||||||
attachCommands(l, c, ssh, ifce)
|
attachCommands(l, c, ssh, ifce)
|
||||||
|
|
||||||
// Start DNS server last to allow using the nebula IP as lighthouse.dns.host
|
|
||||||
var dnsStart func()
|
|
||||||
if lightHouse.amLighthouse && serveDns {
|
|
||||||
l.Debugln("Starting dns server")
|
|
||||||
dnsStart = dnsMain(l, pki.getCertState(), hostMap, c)
|
|
||||||
}
|
|
||||||
|
|
||||||
return &Control{
|
return &Control{
|
||||||
state: StateReady,
|
state: StateReady,
|
||||||
f: ifce,
|
f: ifce,
|
||||||
@@ -295,7 +284,7 @@ func Main(c *config.C, configTest bool, buildVersion string, logger *logrus.Logg
|
|||||||
cancel: cancel,
|
cancel: cancel,
|
||||||
sshStart: sshStart,
|
sshStart: sshStart,
|
||||||
statsStart: statsStart,
|
statsStart: statsStart,
|
||||||
dnsStart: dnsStart,
|
dnsStart: ds.Start,
|
||||||
lighthouseStart: lightHouse.StartUpdateWorker,
|
lighthouseStart: lightHouse.StartUpdateWorker,
|
||||||
connectionManagerStart: connManager.Start,
|
connectionManagerStart: connManager.Start,
|
||||||
}, nil
|
}, nil
|
||||||
|
|||||||
+41
-12
@@ -19,7 +19,7 @@ const (
|
|||||||
minFwPacketLen = 4
|
minFwPacketLen = 4
|
||||||
)
|
)
|
||||||
|
|
||||||
func (f *Interface) readOutsidePackets(via ViaSender, out []byte, packet []byte, h *header.H, fwPacket *firewall.Packet, lhf *LightHouseHandler, nb []byte, q int, localCache firewall.ConntrackCache) {
|
func (f *Interface) readOutsidePackets(via ViaSender, out []byte, packet []byte, h *header.H, fwPacket *firewall.Packet, fwCtx *firewall.PacketContext, lhf *LightHouseHandler, nb []byte, q int, localCache firewall.ConntrackCache) {
|
||||||
err := h.Parse(packet)
|
err := h.Parse(packet)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
// Hole punch packets are 0 or 1 byte big, so lets ignore printing those errors
|
// Hole punch packets are 0 or 1 byte big, so lets ignore printing those errors
|
||||||
@@ -60,7 +60,7 @@ func (f *Interface) readOutsidePackets(via ViaSender, out []byte, packet []byte,
|
|||||||
|
|
||||||
switch h.Subtype {
|
switch h.Subtype {
|
||||||
case header.MessageNone:
|
case header.MessageNone:
|
||||||
if !f.decryptToTun(hostinfo, h.MessageCounter, out, packet, fwPacket, nb, q, localCache) {
|
if !f.decryptToTun(hostinfo, h.MessageCounter, out, packet, fwPacket, fwCtx, nb, q, localCache) {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
case header.MessageRelay:
|
case header.MessageRelay:
|
||||||
@@ -102,7 +102,7 @@ func (f *Interface) readOutsidePackets(via ViaSender, out []byte, packet []byte,
|
|||||||
relay: relay,
|
relay: relay,
|
||||||
IsRelayed: true,
|
IsRelayed: true,
|
||||||
}
|
}
|
||||||
f.readOutsidePackets(via, out[:0], signedPayload, h, fwPacket, lhf, nb, q, localCache)
|
f.readOutsidePackets(via, out[:0], signedPayload, h, fwPacket, fwCtx, lhf, nb, q, localCache)
|
||||||
return
|
return
|
||||||
case ForwardingType:
|
case ForwardingType:
|
||||||
// Find the target HostInfo relay object
|
// Find the target HostInfo relay object
|
||||||
@@ -295,7 +295,10 @@ var (
|
|||||||
)
|
)
|
||||||
|
|
||||||
// newPacket validates and parses the interesting bits for the firewall out of the ip and sub protocol headers
|
// newPacket validates and parses the interesting bits for the firewall out of the ip and sub protocol headers
|
||||||
func newPacket(data []byte, incoming bool, fp *firewall.Packet) error {
|
func newPacket(data []byte, incoming bool, fp *firewall.Packet, ctx *firewall.PacketContext) error {
|
||||||
|
if ctx != nil {
|
||||||
|
*ctx = firewall.PacketContext{}
|
||||||
|
}
|
||||||
if len(data) < 1 {
|
if len(data) < 1 {
|
||||||
return ErrPacketTooShort
|
return ErrPacketTooShort
|
||||||
}
|
}
|
||||||
@@ -303,14 +306,14 @@ func newPacket(data []byte, incoming bool, fp *firewall.Packet) error {
|
|||||||
version := int((data[0] >> 4) & 0x0f)
|
version := int((data[0] >> 4) & 0x0f)
|
||||||
switch version {
|
switch version {
|
||||||
case ipv4.Version:
|
case ipv4.Version:
|
||||||
return parseV4(data, incoming, fp)
|
return parseV4(data, incoming, fp, ctx)
|
||||||
case ipv6.Version:
|
case ipv6.Version:
|
||||||
return parseV6(data, incoming, fp)
|
return parseV6(data, incoming, fp, ctx)
|
||||||
}
|
}
|
||||||
return ErrUnknownIPVersion
|
return ErrUnknownIPVersion
|
||||||
}
|
}
|
||||||
|
|
||||||
func parseV6(data []byte, incoming bool, fp *firewall.Packet) error {
|
func parseV6(data []byte, incoming bool, fp *firewall.Packet, ctx *firewall.PacketContext) error {
|
||||||
dataLen := len(data)
|
dataLen := len(data)
|
||||||
if dataLen < ipv6.HeaderLen {
|
if dataLen < ipv6.HeaderLen {
|
||||||
return ErrIPv6PacketTooShort
|
return ErrIPv6PacketTooShort
|
||||||
@@ -355,6 +358,11 @@ func parseV6(data []byte, incoming bool, fp *firewall.Packet) error {
|
|||||||
fp.RemotePort = 0
|
fp.RemotePort = 0
|
||||||
}
|
}
|
||||||
fp.Fragment = false
|
fp.Fragment = false
|
||||||
|
if ctx != nil {
|
||||||
|
ctx.Length = binary.BigEndian.Uint16(data[4:6]) + uint16(ipv6.HeaderLen)
|
||||||
|
ctx.ICMPType = data[offset]
|
||||||
|
ctx.ICMPCode = data[offset+1]
|
||||||
|
}
|
||||||
return nil
|
return nil
|
||||||
|
|
||||||
case layers.IPProtocolTCP, layers.IPProtocolUDP:
|
case layers.IPProtocolTCP, layers.IPProtocolUDP:
|
||||||
@@ -372,6 +380,12 @@ func parseV6(data []byte, incoming bool, fp *firewall.Packet) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fp.Fragment = false
|
fp.Fragment = false
|
||||||
|
if ctx != nil {
|
||||||
|
ctx.Length = binary.BigEndian.Uint16(data[4:6]) + uint16(ipv6.HeaderLen)
|
||||||
|
if proto == layers.IPProtocolTCP && dataLen >= offset+14 {
|
||||||
|
ctx.TCPFlags = data[offset+13]
|
||||||
|
}
|
||||||
|
}
|
||||||
return nil
|
return nil
|
||||||
|
|
||||||
case layers.IPProtocolIPv6Fragment:
|
case layers.IPProtocolIPv6Fragment:
|
||||||
@@ -423,7 +437,7 @@ func parseV6(data []byte, incoming bool, fp *firewall.Packet) error {
|
|||||||
return ErrIPv6CouldNotFindPayload
|
return ErrIPv6CouldNotFindPayload
|
||||||
}
|
}
|
||||||
|
|
||||||
func parseV4(data []byte, incoming bool, fp *firewall.Packet) error {
|
func parseV4(data []byte, incoming bool, fp *firewall.Packet, ctx *firewall.PacketContext) error {
|
||||||
// Do we at least have an ipv4 header worth of data?
|
// Do we at least have an ipv4 header worth of data?
|
||||||
if len(data) < ipv4.HeaderLen {
|
if len(data) < ipv4.HeaderLen {
|
||||||
return ErrIPv4PacketTooShort
|
return ErrIPv4PacketTooShort
|
||||||
@@ -480,6 +494,21 @@ func parseV4(data []byte, incoming bool, fp *firewall.Packet) error {
|
|||||||
fp.RemotePort = binary.BigEndian.Uint16(data[ihl+2 : ihl+4]) //dst port
|
fp.RemotePort = binary.BigEndian.Uint16(data[ihl+2 : ihl+4]) //dst port
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if ctx != nil {
|
||||||
|
ctx.Length = binary.BigEndian.Uint16(data[2:4])
|
||||||
|
if !fp.Fragment {
|
||||||
|
switch fp.Protocol {
|
||||||
|
case firewall.ProtoICMP:
|
||||||
|
ctx.ICMPType = data[ihl]
|
||||||
|
ctx.ICMPCode = data[ihl+1]
|
||||||
|
case firewall.ProtoTCP:
|
||||||
|
if len(data) >= ihl+14 {
|
||||||
|
ctx.TCPFlags = data[ihl+13]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -499,7 +528,7 @@ func (f *Interface) decrypt(hostinfo *HostInfo, mc uint64, out []byte, packet []
|
|||||||
return out, nil
|
return out, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *Interface) decryptToTun(hostinfo *HostInfo, messageCounter uint64, out []byte, packet []byte, fwPacket *firewall.Packet, nb []byte, q int, localCache firewall.ConntrackCache) bool {
|
func (f *Interface) decryptToTun(hostinfo *HostInfo, messageCounter uint64, out []byte, packet []byte, fwPacket *firewall.Packet, fwCtx *firewall.PacketContext, nb []byte, q int, localCache firewall.ConntrackCache) bool {
|
||||||
var err error
|
var err error
|
||||||
|
|
||||||
out, err = hostinfo.ConnectionState.dKey.DecryptDanger(out, packet[:header.Len], packet[header.Len:], messageCounter, nb)
|
out, err = hostinfo.ConnectionState.dKey.DecryptDanger(out, packet[:header.Len], packet[header.Len:], messageCounter, nb)
|
||||||
@@ -508,7 +537,7 @@ func (f *Interface) decryptToTun(hostinfo *HostInfo, messageCounter uint64, out
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
err = newPacket(out, true, fwPacket)
|
err = newPacket(out, true, fwPacket, fwCtx)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
hostinfo.logger(f.l).WithError(err).WithField("packet", out).
|
hostinfo.logger(f.l).WithError(err).WithField("packet", out).
|
||||||
Warnf("Error while validating inbound packet")
|
Warnf("Error while validating inbound packet")
|
||||||
@@ -521,7 +550,7 @@ func (f *Interface) decryptToTun(hostinfo *HostInfo, messageCounter uint64, out
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
dropReason := f.firewall.Drop(*fwPacket, true, hostinfo, f.pki.GetCAPool(), localCache)
|
dropReason := f.firewall.Drop(*fwPacket, *fwCtx, true, hostinfo, f.pki.GetCAPool(), localCache)
|
||||||
if dropReason != nil {
|
if dropReason != nil {
|
||||||
// NOTE: We give `packet` as the `out` here since we already decrypted from it and we don't need it anymore
|
// NOTE: We give `packet` as the `out` here since we already decrypted from it and we don't need it anymore
|
||||||
// This gives us a buffer to build the reject packet in
|
// This gives us a buffer to build the reject packet in
|
||||||
@@ -535,7 +564,7 @@ func (f *Interface) decryptToTun(hostinfo *HostInfo, messageCounter uint64, out
|
|||||||
}
|
}
|
||||||
|
|
||||||
f.connectionManager.In(hostinfo)
|
f.connectionManager.In(hostinfo)
|
||||||
err = f.tunCoalescers[q].Add(out)
|
_, err = f.readers[q].Write(out)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
f.l.WithError(err).Error("Failed to write to tun")
|
f.l.WithError(err).Error("Failed to write to tun")
|
||||||
}
|
}
|
||||||
|
|||||||
+34
-34
@@ -20,13 +20,13 @@ func Test_newPacket(t *testing.T) {
|
|||||||
p := &firewall.Packet{}
|
p := &firewall.Packet{}
|
||||||
|
|
||||||
// length fails
|
// length fails
|
||||||
err := newPacket([]byte{}, true, p)
|
err := newPacket([]byte{}, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrPacketTooShort)
|
require.ErrorIs(t, err, ErrPacketTooShort)
|
||||||
|
|
||||||
err = newPacket([]byte{0x40}, true, p)
|
err = newPacket([]byte{0x40}, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv4PacketTooShort)
|
require.ErrorIs(t, err, ErrIPv4PacketTooShort)
|
||||||
|
|
||||||
err = newPacket([]byte{0x60}, true, p)
|
err = newPacket([]byte{0x60}, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||||
|
|
||||||
// length fail with ip options
|
// length fail with ip options
|
||||||
@@ -39,15 +39,15 @@ func Test_newPacket(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
b, _ := h.Marshal()
|
b, _ := h.Marshal()
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv4InvalidHeaderLength)
|
require.ErrorIs(t, err, ErrIPv4InvalidHeaderLength)
|
||||||
|
|
||||||
// not an ipv4 packet
|
// not an ipv4 packet
|
||||||
err = newPacket([]byte{0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}, true, p)
|
err = newPacket([]byte{0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrUnknownIPVersion)
|
require.ErrorIs(t, err, ErrUnknownIPVersion)
|
||||||
|
|
||||||
// invalid ihl
|
// invalid ihl
|
||||||
err = newPacket([]byte{4<<4 | (8 >> 2 & 0x0f), 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}, true, p)
|
err = newPacket([]byte{4<<4 | (8 >> 2 & 0x0f), 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv4InvalidHeaderLength)
|
require.ErrorIs(t, err, ErrIPv4InvalidHeaderLength)
|
||||||
|
|
||||||
// account for variable ip header length - incoming
|
// account for variable ip header length - incoming
|
||||||
@@ -62,7 +62,7 @@ func Test_newPacket(t *testing.T) {
|
|||||||
|
|
||||||
b, _ = h.Marshal()
|
b, _ = h.Marshal()
|
||||||
b = append(b, []byte{0, 3, 0, 4}...)
|
b = append(b, []byte{0, 3, 0, 4}...)
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
|
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(firewall.ProtoTCP), p.Protocol)
|
assert.Equal(t, uint8(firewall.ProtoTCP), p.Protocol)
|
||||||
@@ -84,7 +84,7 @@ func Test_newPacket(t *testing.T) {
|
|||||||
|
|
||||||
b, _ = h.Marshal()
|
b, _ = h.Marshal()
|
||||||
b = append(b, []byte{0, 5, 0, 6}...)
|
b = append(b, []byte{0, 5, 0, 6}...)
|
||||||
err = newPacket(b, false, p)
|
err = newPacket(b, false, p, nil)
|
||||||
|
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(2), p.Protocol)
|
assert.Equal(t, uint8(2), p.Protocol)
|
||||||
@@ -114,7 +114,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
err := gopacket.SerializeLayers(buffer, opt, &ip)
|
err := gopacket.SerializeLayers(buffer, opt, &ip)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
|
|
||||||
err = newPacket(buffer.Bytes(), true, p)
|
err = newPacket(buffer.Bytes(), true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||||
|
|
||||||
// A v6 packet with a hop-by-hop extension
|
// A v6 packet with a hop-by-hop extension
|
||||||
@@ -148,12 +148,12 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
|
|
||||||
// A full IPv6 header and 1 byte in the first extension, but missing
|
// A full IPv6 header and 1 byte in the first extension, but missing
|
||||||
// the length byte.
|
// the length byte.
|
||||||
err = newPacket(buffer.Bytes()[:41], true, p)
|
err = newPacket(buffer.Bytes()[:41], true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||||
|
|
||||||
// A full IPv6 header plus 1 full extension, but only 1 byte of the
|
// A full IPv6 header plus 1 full extension, but only 1 byte of the
|
||||||
// next layer, missing length byte
|
// next layer, missing length byte
|
||||||
err = newPacket(buffer.Bytes()[:49], true, p)
|
err = newPacket(buffer.Bytes()[:49], true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||||
err = nil
|
err = nil
|
||||||
|
|
||||||
@@ -173,7 +173,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
|
|
||||||
buffer.Clear()
|
buffer.Clear()
|
||||||
require.NoError(t, gopacket.SerializeLayers(buffer, opt, &ip, &icmp))
|
require.NoError(t, gopacket.SerializeLayers(buffer, opt, &ip, &icmp))
|
||||||
require.Error(t, newPacket(buffer.Bytes(), true, p))
|
require.Error(t, newPacket(buffer.Bytes(), true, p, nil))
|
||||||
|
|
||||||
buffer.Clear()
|
buffer.Clear()
|
||||||
echo := layers.ICMPv6Echo{
|
echo := layers.ICMPv6Echo{
|
||||||
@@ -181,7 +181,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
SeqNumber: 1234,
|
SeqNumber: 1234,
|
||||||
}
|
}
|
||||||
require.NoError(t, gopacket.SerializeLayers(buffer, opt, &ip, &icmp, &echo))
|
require.NoError(t, gopacket.SerializeLayers(buffer, opt, &ip, &icmp, &echo))
|
||||||
require.NoError(t, newPacket(buffer.Bytes(), true, p))
|
require.NoError(t, newPacket(buffer.Bytes(), true, p, nil))
|
||||||
assert.Equal(t, uint8(layers.IPProtocolICMPv6), p.Protocol)
|
assert.Equal(t, uint8(layers.IPProtocolICMPv6), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.LocalAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.LocalAddr)
|
||||||
@@ -192,7 +192,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
// A good ESP packet
|
// A good ESP packet
|
||||||
b := buffer.Bytes()
|
b := buffer.Bytes()
|
||||||
b[6] = byte(layers.IPProtocolESP)
|
b[6] = byte(layers.IPProtocolESP)
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(layers.IPProtocolESP), p.Protocol)
|
assert.Equal(t, uint8(layers.IPProtocolESP), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
@@ -204,7 +204,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
// A good None packet
|
// A good None packet
|
||||||
b = buffer.Bytes()
|
b = buffer.Bytes()
|
||||||
b[6] = byte(layers.IPProtocolNoNextHeader)
|
b[6] = byte(layers.IPProtocolNoNextHeader)
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(layers.IPProtocolNoNextHeader), p.Protocol)
|
assert.Equal(t, uint8(layers.IPProtocolNoNextHeader), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
@@ -216,7 +216,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
// An unknown protocol packet
|
// An unknown protocol packet
|
||||||
b = buffer.Bytes()
|
b = buffer.Bytes()
|
||||||
b[6] = 255 // 255 is a reserved protocol number
|
b[6] = 255 // 255 is a reserved protocol number
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||||
|
|
||||||
// A good UDP packet
|
// A good UDP packet
|
||||||
@@ -243,7 +243,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
b = buffer.Bytes()
|
b = buffer.Bytes()
|
||||||
|
|
||||||
// incoming
|
// incoming
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(firewall.ProtoUDP), p.Protocol)
|
assert.Equal(t, uint8(firewall.ProtoUDP), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
@@ -253,7 +253,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
assert.False(t, p.Fragment)
|
assert.False(t, p.Fragment)
|
||||||
|
|
||||||
// outgoing
|
// outgoing
|
||||||
err = newPacket(b, false, p)
|
err = newPacket(b, false, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(firewall.ProtoUDP), p.Protocol)
|
assert.Equal(t, uint8(firewall.ProtoUDP), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
||||||
@@ -263,14 +263,14 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
assert.False(t, p.Fragment)
|
assert.False(t, p.Fragment)
|
||||||
|
|
||||||
// Too short UDP packet
|
// Too short UDP packet
|
||||||
err = newPacket(b[:len(b)-10], false, p) // pull off the last 10 bytes
|
err = newPacket(b[:len(b)-10], false, p, nil) // pull off the last 10 bytes
|
||||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||||
|
|
||||||
// A good TCP packet
|
// A good TCP packet
|
||||||
b[6] = byte(layers.IPProtocolTCP)
|
b[6] = byte(layers.IPProtocolTCP)
|
||||||
|
|
||||||
// incoming
|
// incoming
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(firewall.ProtoTCP), p.Protocol)
|
assert.Equal(t, uint8(firewall.ProtoTCP), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
@@ -280,7 +280,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
assert.False(t, p.Fragment)
|
assert.False(t, p.Fragment)
|
||||||
|
|
||||||
// outgoing
|
// outgoing
|
||||||
err = newPacket(b, false, p)
|
err = newPacket(b, false, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(firewall.ProtoTCP), p.Protocol)
|
assert.Equal(t, uint8(firewall.ProtoTCP), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
||||||
@@ -290,7 +290,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
assert.False(t, p.Fragment)
|
assert.False(t, p.Fragment)
|
||||||
|
|
||||||
// Too short TCP packet
|
// Too short TCP packet
|
||||||
err = newPacket(b[:len(b)-10], false, p) // pull off the last 10 bytes
|
err = newPacket(b[:len(b)-10], false, p, nil) // pull off the last 10 bytes
|
||||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||||
|
|
||||||
// A good UDP packet with an AH header
|
// A good UDP packet with an AH header
|
||||||
@@ -325,7 +325,7 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
b = append(b, ahb...)
|
b = append(b, ahb...)
|
||||||
b = append(b, udpHeader...)
|
b = append(b, udpHeader...)
|
||||||
|
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, uint8(firewall.ProtoUDP), p.Protocol)
|
assert.Equal(t, uint8(firewall.ProtoUDP), p.Protocol)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
@@ -335,12 +335,12 @@ func Test_newPacket_v6(t *testing.T) {
|
|||||||
assert.False(t, p.Fragment)
|
assert.False(t, p.Fragment)
|
||||||
|
|
||||||
// Ensure buffer bounds checking during processing
|
// Ensure buffer bounds checking during processing
|
||||||
err = newPacket(b[:41], true, p)
|
err = newPacket(b[:41], true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||||
|
|
||||||
// Invalid AH header
|
// Invalid AH header
|
||||||
b = buffer.Bytes()
|
b = buffer.Bytes()
|
||||||
err = newPacket(b, true, p)
|
err = newPacket(b, true, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -388,7 +388,7 @@ func Test_newPacket_ipv6Fragment(t *testing.T) {
|
|||||||
firstFrag = append(firstFrag, []byte{0xde, 0xad, 0xbe, 0xef}...)
|
firstFrag = append(firstFrag, []byte{0xde, 0xad, 0xbe, 0xef}...)
|
||||||
|
|
||||||
// Test first fragment incoming
|
// Test first fragment incoming
|
||||||
err = newPacket(firstFrag, true, p)
|
err = newPacket(firstFrag, true, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.LocalAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.LocalAddr)
|
||||||
@@ -398,7 +398,7 @@ func Test_newPacket_ipv6Fragment(t *testing.T) {
|
|||||||
assert.False(t, p.Fragment)
|
assert.False(t, p.Fragment)
|
||||||
|
|
||||||
// Test first fragment outgoing
|
// Test first fragment outgoing
|
||||||
err = newPacket(firstFrag, false, p)
|
err = newPacket(firstFrag, false, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.RemoteAddr)
|
||||||
@@ -427,7 +427,7 @@ func Test_newPacket_ipv6Fragment(t *testing.T) {
|
|||||||
secondFrag = append(secondFrag, []byte{0xde, 0xad, 0xbe, 0xef}...)
|
secondFrag = append(secondFrag, []byte{0xde, 0xad, 0xbe, 0xef}...)
|
||||||
|
|
||||||
// Test second fragment incoming
|
// Test second fragment incoming
|
||||||
err = newPacket(secondFrag, true, p)
|
err = newPacket(secondFrag, true, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.RemoteAddr)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.LocalAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.LocalAddr)
|
||||||
@@ -437,7 +437,7 @@ func Test_newPacket_ipv6Fragment(t *testing.T) {
|
|||||||
assert.True(t, p.Fragment)
|
assert.True(t, p.Fragment)
|
||||||
|
|
||||||
// Test second fragment outgoing
|
// Test second fragment outgoing
|
||||||
err = newPacket(secondFrag, false, p)
|
err = newPacket(secondFrag, false, p, nil)
|
||||||
require.NoError(t, err)
|
require.NoError(t, err)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::2"), p.LocalAddr)
|
||||||
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.RemoteAddr)
|
assert.Equal(t, netip.MustParseAddr("ff02::1"), p.RemoteAddr)
|
||||||
@@ -447,7 +447,7 @@ func Test_newPacket_ipv6Fragment(t *testing.T) {
|
|||||||
assert.True(t, p.Fragment)
|
assert.True(t, p.Fragment)
|
||||||
|
|
||||||
// Too short of a fragment packet
|
// Too short of a fragment packet
|
||||||
err = newPacket(secondFrag[:len(secondFrag)-10], false, p)
|
err = newPacket(secondFrag[:len(secondFrag)-10], false, p, nil)
|
||||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -529,7 +529,7 @@ func BenchmarkParseV6(b *testing.B) {
|
|||||||
|
|
||||||
b.Run("Normal", func(b *testing.B) {
|
b.Run("Normal", func(b *testing.B) {
|
||||||
for i := 0; i < b.N; i++ {
|
for i := 0; i < b.N; i++ {
|
||||||
if err = parseV6(normalPacket, true, fp); err != nil {
|
if err = parseV6(normalPacket, true, fp, nil); err != nil {
|
||||||
b.Fatal(err)
|
b.Fatal(err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -537,7 +537,7 @@ func BenchmarkParseV6(b *testing.B) {
|
|||||||
|
|
||||||
b.Run("FirstFragment", func(b *testing.B) {
|
b.Run("FirstFragment", func(b *testing.B) {
|
||||||
for i := 0; i < b.N; i++ {
|
for i := 0; i < b.N; i++ {
|
||||||
if err = parseV6(firstFrag, true, fp); err != nil {
|
if err = parseV6(firstFrag, true, fp, nil); err != nil {
|
||||||
b.Fatal(err)
|
b.Fatal(err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -545,7 +545,7 @@ func BenchmarkParseV6(b *testing.B) {
|
|||||||
|
|
||||||
b.Run("SecondFragment", func(b *testing.B) {
|
b.Run("SecondFragment", func(b *testing.B) {
|
||||||
for i := 0; i < b.N; i++ {
|
for i := 0; i < b.N; i++ {
|
||||||
if err = parseV6(secondFrag, true, fp); err != nil {
|
if err = parseV6(secondFrag, true, fp, nil); err != nil {
|
||||||
b.Fatal(err)
|
b.Fatal(err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -590,7 +590,7 @@ func BenchmarkParseV6(b *testing.B) {
|
|||||||
|
|
||||||
b.Run("200 HopByHop headers", func(b *testing.B) {
|
b.Run("200 HopByHop headers", func(b *testing.B) {
|
||||||
for i := 0; i < b.N; i++ {
|
for i := 0; i < b.N; i++ {
|
||||||
if err = parseV6(evilBytes, false, fp); err != nil {
|
if err = parseV6(evilBytes, false, fp, nil); err != nil {
|
||||||
b.Fatal(err)
|
b.Fatal(err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-53
@@ -7,63 +7,12 @@ import (
|
|||||||
"github.com/slackhq/nebula/routing"
|
"github.com/slackhq/nebula/routing"
|
||||||
)
|
)
|
||||||
|
|
||||||
// defaultBatchBufSize is the per-Queue scratch size for Read on backends
|
|
||||||
// that don't do TSO segmentation. 65535 covers any single IP packet.
|
|
||||||
const defaultBatchBufSize = 65535
|
|
||||||
|
|
||||||
// Queue is a readable/writable tun queue. One Queue is driven by a single
|
|
||||||
// read goroutine plus concurrent writers (see Write / WriteReject below).
|
|
||||||
type Queue interface {
|
|
||||||
io.Closer
|
|
||||||
|
|
||||||
// Read returns one or more packets. The returned slices are borrowed
|
|
||||||
// from the Queue's internal buffer and are only valid until the next
|
|
||||||
// Read or Close on this Queue — callers must encrypt or copy each
|
|
||||||
// slice before the next call. Not safe for concurrent Reads; exactly
|
|
||||||
// one goroutine per Queue reads.
|
|
||||||
Read() ([][]byte, error)
|
|
||||||
|
|
||||||
// Write emits a single packet on the plaintext (outside→inside)
|
|
||||||
// delivery path. May run concurrently with WriteReject on the same
|
|
||||||
// Queue, but not with itself.
|
|
||||||
Write(p []byte) (int, error)
|
|
||||||
|
|
||||||
// WriteReject writes a single packet that originated from the inside
|
|
||||||
// path (reject replies or self-forward) using scratch state distinct
|
|
||||||
// from Write, so it can run concurrently with Write on the same Queue
|
|
||||||
// without a data race. On backends without a shared-scratch Write, a
|
|
||||||
// trivial delegation to Write is acceptable.
|
|
||||||
WriteReject(p []byte) (int, error)
|
|
||||||
}
|
|
||||||
|
|
||||||
type Device interface {
|
type Device interface {
|
||||||
Queue
|
io.ReadWriteCloser
|
||||||
Activate() error
|
Activate() error
|
||||||
Networks() []netip.Prefix
|
Networks() []netip.Prefix
|
||||||
Name() string
|
Name() string
|
||||||
RoutesFor(netip.Addr) routing.Gateways
|
RoutesFor(netip.Addr) routing.Gateways
|
||||||
SupportsMultiqueue() bool
|
SupportsMultiqueue() bool
|
||||||
NewMultiQueueReader() (Queue, error)
|
NewMultiQueueReader() (io.ReadWriteCloser, error)
|
||||||
}
|
|
||||||
|
|
||||||
// GSOWriter is implemented by Queues that can emit a TCP TSO superpacket
|
|
||||||
// assembled from a header prefix plus one or more borrowed payload
|
|
||||||
// fragments, in a single vectored write (writev with a leading
|
|
||||||
// virtio_net_hdr). This lets the coalescer avoid copying payload bytes
|
|
||||||
// between the caller's decrypt buffer and the TUN. Backends without GSO
|
|
||||||
// support return false from GSOSupported and coalescing is skipped.
|
|
||||||
//
|
|
||||||
// hdr contains the IPv4/IPv6 + TCP header prefix (mutable — callers will
|
|
||||||
// have filled in total length and pseudo-header partial). pays are
|
|
||||||
// non-overlapping payload fragments whose concatenation is the full
|
|
||||||
// superpacket payload; they are read-only from the writer's perspective
|
|
||||||
// and must remain valid until the call returns. gsoSize is the MSS:
|
|
||||||
// every segment except possibly the last is exactly that many bytes.
|
|
||||||
// csumStart is the byte offset where the TCP header begins within hdr.
|
|
||||||
//
|
|
||||||
// hdr's TCP checksum field must already hold the pseudo-header partial
|
|
||||||
// sum (single-fold, not inverted), per virtio NEEDS_CSUM semantics.
|
|
||||||
type GSOWriter interface {
|
|
||||||
WriteGSO(hdr []byte, pays [][]byte, gsoSize uint16, isV6 bool, csumStart uint16) error
|
|
||||||
GSOSupported() bool
|
|
||||||
}
|
}
|
||||||
|
|||||||
+6
-33
@@ -18,39 +18,12 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
type tun struct {
|
type tun struct {
|
||||||
rwc io.ReadWriteCloser
|
io.ReadWriteCloser
|
||||||
fd int
|
fd int
|
||||||
vpnNetworks []netip.Prefix
|
vpnNetworks []netip.Prefix
|
||||||
Routes atomic.Pointer[[]Route]
|
Routes atomic.Pointer[[]Route]
|
||||||
routeTree atomic.Pointer[bart.Table[routing.Gateways]]
|
routeTree atomic.Pointer[bart.Table[routing.Gateways]]
|
||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Read() ([][]byte, error) {
|
|
||||||
if t.readBuf == nil {
|
|
||||||
t.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := t.rwc.Read(t.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t.batchRet[0] = t.readBuf[:n]
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Write(p []byte) (int, error) {
|
|
||||||
return t.rwc.Write(p)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) WriteReject(p []byte) (int, error) {
|
|
||||||
return t.rwc.Write(p)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Close() error {
|
|
||||||
return t.rwc.Close()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []netip.Prefix) (*tun, error) {
|
func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []netip.Prefix) (*tun, error) {
|
||||||
@@ -59,10 +32,10 @@ func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []net
|
|||||||
file := os.NewFile(uintptr(deviceFd), "/dev/net/tun")
|
file := os.NewFile(uintptr(deviceFd), "/dev/net/tun")
|
||||||
|
|
||||||
t := &tun{
|
t := &tun{
|
||||||
rwc: file,
|
ReadWriteCloser: file,
|
||||||
fd: deviceFd,
|
fd: deviceFd,
|
||||||
vpnNetworks: vpnNetworks,
|
vpnNetworks: vpnNetworks,
|
||||||
l: l,
|
l: l,
|
||||||
}
|
}
|
||||||
|
|
||||||
err := t.reload(c, true)
|
err := t.reload(c, true)
|
||||||
@@ -126,6 +99,6 @@ func (t *tun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) NewMultiQueueReader() (Queue, error) {
|
func (t *tun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented for android")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented for android")
|
||||||
}
|
}
|
||||||
|
|||||||
+12
-31
@@ -23,7 +23,7 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
type tun struct {
|
type tun struct {
|
||||||
rwc io.ReadWriteCloser
|
io.ReadWriteCloser
|
||||||
Device string
|
Device string
|
||||||
vpnNetworks []netip.Prefix
|
vpnNetworks []netip.Prefix
|
||||||
DefaultMTU int
|
DefaultMTU int
|
||||||
@@ -34,9 +34,6 @@ type tun struct {
|
|||||||
|
|
||||||
// cache out buffer since we need to prepend 4 bytes for tun metadata
|
// cache out buffer since we need to prepend 4 bytes for tun metadata
|
||||||
out []byte
|
out []byte
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
}
|
||||||
|
|
||||||
type ifReq struct {
|
type ifReq struct {
|
||||||
@@ -127,11 +124,11 @@ func newTun(c *config.C, l *logrus.Logger, vpnNetworks []netip.Prefix, _ bool) (
|
|||||||
}
|
}
|
||||||
|
|
||||||
t := &tun{
|
t := &tun{
|
||||||
rwc: os.NewFile(uintptr(fd), ""),
|
ReadWriteCloser: os.NewFile(uintptr(fd), ""),
|
||||||
Device: name,
|
Device: name,
|
||||||
vpnNetworks: vpnNetworks,
|
vpnNetworks: vpnNetworks,
|
||||||
DefaultMTU: c.GetInt("tun.mtu", DefaultMTU),
|
DefaultMTU: c.GetInt("tun.mtu", DefaultMTU),
|
||||||
l: l,
|
l: l,
|
||||||
}
|
}
|
||||||
|
|
||||||
err = t.reload(c, true)
|
err = t.reload(c, true)
|
||||||
@@ -161,8 +158,8 @@ func newTunFromFd(_ *config.C, _ *logrus.Logger, _ int, _ []netip.Prefix) (*tun,
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) Close() error {
|
func (t *tun) Close() error {
|
||||||
if t.rwc != nil {
|
if t.ReadWriteCloser != nil {
|
||||||
return t.rwc.Close()
|
return t.ReadWriteCloser.Close()
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -506,31 +503,15 @@ func delRoute(prefix netip.Prefix, gateway netroute.Addr) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) readOne(to []byte) (int, error) {
|
func (t *tun) Read(to []byte) (int, error) {
|
||||||
buf := make([]byte, len(to)+4)
|
buf := make([]byte, len(to)+4)
|
||||||
|
|
||||||
n, err := t.rwc.Read(buf)
|
n, err := t.ReadWriteCloser.Read(buf)
|
||||||
|
|
||||||
copy(to, buf[4:])
|
copy(to, buf[4:])
|
||||||
return n - 4, err
|
return n - 4, err
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) Read() ([][]byte, error) {
|
|
||||||
if t.readBuf == nil {
|
|
||||||
t.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := t.readOne(t.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t.batchRet[0] = t.readBuf[:n]
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) WriteReject(p []byte) (int, error) {
|
|
||||||
return t.Write(p)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Write is only valid for single threaded use
|
// Write is only valid for single threaded use
|
||||||
func (t *tun) Write(from []byte) (int, error) {
|
func (t *tun) Write(from []byte) (int, error) {
|
||||||
buf := t.out
|
buf := t.out
|
||||||
@@ -556,7 +537,7 @@ func (t *tun) Write(from []byte) (int, error) {
|
|||||||
|
|
||||||
copy(buf[4:], from)
|
copy(buf[4:], from)
|
||||||
|
|
||||||
n, err := t.rwc.Write(buf)
|
n, err := t.ReadWriteCloser.Write(buf)
|
||||||
return n - 4, err
|
return n - 4, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -572,6 +553,6 @@ func (t *tun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) NewMultiQueueReader() (Queue, error) {
|
func (t *tun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented for darwin")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented for darwin")
|
||||||
}
|
}
|
||||||
|
|||||||
+19
-22
@@ -20,23 +20,6 @@ type disabledTun struct {
|
|||||||
tx metrics.Counter
|
tx metrics.Counter
|
||||||
rx metrics.Counter
|
rx metrics.Counter
|
||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *disabledTun) Read() ([][]byte, error) {
|
|
||||||
r, ok := <-t.read
|
|
||||||
if !ok {
|
|
||||||
return nil, io.EOF
|
|
||||||
}
|
|
||||||
|
|
||||||
t.tx.Inc(1)
|
|
||||||
if t.l.Level >= logrus.DebugLevel {
|
|
||||||
t.l.WithField("raw", prettyPacket(r)).Debugf("Write payload")
|
|
||||||
}
|
|
||||||
|
|
||||||
t.batchRet[0] = r
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func newDisabledTun(vpnNetworks []netip.Prefix, queueLen int, metricsEnabled bool, l *logrus.Logger) *disabledTun {
|
func newDisabledTun(vpnNetworks []netip.Prefix, queueLen int, metricsEnabled bool, l *logrus.Logger) *disabledTun {
|
||||||
@@ -73,6 +56,24 @@ func (*disabledTun) Name() string {
|
|||||||
return "disabled"
|
return "disabled"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (t *disabledTun) Read(b []byte) (int, error) {
|
||||||
|
r, ok := <-t.read
|
||||||
|
if !ok {
|
||||||
|
return 0, io.EOF
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(r) > len(b) {
|
||||||
|
return 0, fmt.Errorf("packet larger than mtu: %d > %d bytes", len(r), len(b))
|
||||||
|
}
|
||||||
|
|
||||||
|
t.tx.Inc(1)
|
||||||
|
if t.l.Level >= logrus.DebugLevel {
|
||||||
|
t.l.WithField("raw", prettyPacket(r)).Debugf("Write payload")
|
||||||
|
}
|
||||||
|
|
||||||
|
return copy(b, r), nil
|
||||||
|
}
|
||||||
|
|
||||||
func (t *disabledTun) handleICMPEchoRequest(b []byte) bool {
|
func (t *disabledTun) handleICMPEchoRequest(b []byte) bool {
|
||||||
out := make([]byte, len(b))
|
out := make([]byte, len(b))
|
||||||
out = iputil.CreateICMPEchoResponse(b, out)
|
out = iputil.CreateICMPEchoResponse(b, out)
|
||||||
@@ -104,15 +105,11 @@ func (t *disabledTun) Write(b []byte) (int, error) {
|
|||||||
return len(b), nil
|
return len(b), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *disabledTun) WriteReject(b []byte) (int, error) {
|
|
||||||
return t.Write(b)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *disabledTun) SupportsMultiqueue() bool {
|
func (t *disabledTun) SupportsMultiqueue() bool {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *disabledTun) NewMultiQueueReader() (Queue, error) {
|
func (t *disabledTun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return t, nil
|
return t, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+3
-21
@@ -7,6 +7,7 @@ import (
|
|||||||
"bytes"
|
"bytes"
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"os"
|
"os"
|
||||||
@@ -100,9 +101,6 @@ type tun struct {
|
|||||||
readPoll [2]unix.PollFd
|
readPoll [2]unix.PollFd
|
||||||
writePoll [2]unix.PollFd
|
writePoll [2]unix.PollFd
|
||||||
closed atomic.Bool
|
closed atomic.Bool
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// blockOnRead waits until the tun fd is readable or shutdown has been signaled.
|
// blockOnRead waits until the tun fd is readable or shutdown has been signaled.
|
||||||
@@ -157,23 +155,7 @@ func (t *tun) blockOnWrite() error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) Read() ([][]byte, error) {
|
func (t *tun) Read(to []byte) (int, error) {
|
||||||
if t.readBuf == nil {
|
|
||||||
t.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := t.readOne(t.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t.batchRet[0] = t.readBuf[:n]
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) WriteReject(p []byte) (int, error) {
|
|
||||||
return t.Write(p)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) readOne(to []byte) (int, error) {
|
|
||||||
// first 4 bytes is protocol family, in network byte order
|
// first 4 bytes is protocol family, in network byte order
|
||||||
var head [4]byte
|
var head [4]byte
|
||||||
iovecs := [2]syscall.Iovec{
|
iovecs := [2]syscall.Iovec{
|
||||||
@@ -581,7 +563,7 @@ func (t *tun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) NewMultiQueueReader() (Queue, error) {
|
func (t *tun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented for freebsd")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented for freebsd")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+5
-32
@@ -21,38 +21,11 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
type tun struct {
|
type tun struct {
|
||||||
rwc io.ReadWriteCloser
|
io.ReadWriteCloser
|
||||||
vpnNetworks []netip.Prefix
|
vpnNetworks []netip.Prefix
|
||||||
Routes atomic.Pointer[[]Route]
|
Routes atomic.Pointer[[]Route]
|
||||||
routeTree atomic.Pointer[bart.Table[routing.Gateways]]
|
routeTree atomic.Pointer[bart.Table[routing.Gateways]]
|
||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Read() ([][]byte, error) {
|
|
||||||
if t.readBuf == nil {
|
|
||||||
t.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := t.rwc.Read(t.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t.batchRet[0] = t.readBuf[:n]
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Write(p []byte) (int, error) {
|
|
||||||
return t.rwc.Write(p)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) WriteReject(p []byte) (int, error) {
|
|
||||||
return t.rwc.Write(p)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Close() error {
|
|
||||||
return t.rwc.Close()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func newTun(_ *config.C, _ *logrus.Logger, _ []netip.Prefix, _ bool) (*tun, error) {
|
func newTun(_ *config.C, _ *logrus.Logger, _ []netip.Prefix, _ bool) (*tun, error) {
|
||||||
@@ -62,9 +35,9 @@ func newTun(_ *config.C, _ *logrus.Logger, _ []netip.Prefix, _ bool) (*tun, erro
|
|||||||
func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []netip.Prefix) (*tun, error) {
|
func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []netip.Prefix) (*tun, error) {
|
||||||
file := os.NewFile(uintptr(deviceFd), "/dev/tun")
|
file := os.NewFile(uintptr(deviceFd), "/dev/tun")
|
||||||
t := &tun{
|
t := &tun{
|
||||||
vpnNetworks: vpnNetworks,
|
vpnNetworks: vpnNetworks,
|
||||||
rwc: &tunReadCloser{f: file},
|
ReadWriteCloser: &tunReadCloser{f: file},
|
||||||
l: l,
|
l: l,
|
||||||
}
|
}
|
||||||
|
|
||||||
err := t.reload(c, true)
|
err := t.reload(c, true)
|
||||||
@@ -182,6 +155,6 @@ func (t *tun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) NewMultiQueueReader() (Queue, error) {
|
func (t *tun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented for ios")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented for ios")
|
||||||
}
|
}
|
||||||
|
|||||||
+53
-383
@@ -10,11 +10,9 @@ import (
|
|||||||
"net"
|
"net"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"os"
|
"os"
|
||||||
"runtime"
|
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
"syscall"
|
|
||||||
"time"
|
"time"
|
||||||
"unsafe"
|
"unsafe"
|
||||||
|
|
||||||
@@ -36,58 +34,16 @@ type tunFile struct {
|
|||||||
readPoll [2]unix.PollFd
|
readPoll [2]unix.PollFd
|
||||||
writePoll [2]unix.PollFd
|
writePoll [2]unix.PollFd
|
||||||
closed bool
|
closed bool
|
||||||
|
|
||||||
// vnetHdr is true when this fd was opened with IFF_VNET_HDR and the
|
|
||||||
// kernel successfully accepted TUNSETOFFLOAD. Reads include a leading
|
|
||||||
// virtio_net_hdr and may carry a TSO superpacket we must segment;
|
|
||||||
// writes must prepend a zeroed virtio_net_hdr.
|
|
||||||
vnetHdr bool
|
|
||||||
readBuf []byte // scratch for a single raw read (virtio hdr + superpacket)
|
|
||||||
segBuf []byte // backing store for segmented output
|
|
||||||
segOff int // cursor into segBuf for the current Read drain
|
|
||||||
pending [][]byte // segments returned from the most recent Read
|
|
||||||
writeIovs [2]unix.Iovec // preallocated iovecs for Write (coalescer passthrough); iovs[0] is fixed to validVnetHdr
|
|
||||||
// rejectIovs is a second preallocated iovec scratch used exclusively by
|
|
||||||
// WriteReject (reject + self-forward from the inside path). It mirrors
|
|
||||||
// writeIovs but lets listenIn goroutines emit reject packets without
|
|
||||||
// racing with the listenOut coalescer that owns writeIovs.
|
|
||||||
rejectIovs [2]unix.Iovec
|
|
||||||
|
|
||||||
// gsoHdrBuf is a per-queue 10-byte scratch for the virtio_net_hdr emitted
|
|
||||||
// by WriteGSO. Separate from validVnetHdr so a concurrent non-GSO Write on
|
|
||||||
// another queue never observes a half-written header.
|
|
||||||
gsoHdrBuf [virtioNetHdrLen]byte
|
|
||||||
// gsoIovs is the writev iovec scratch for WriteGSO. Sized to hold the
|
|
||||||
// virtio header + IP/TCP header + up to gsoInitialPayIovs payload
|
|
||||||
// fragments; grown on demand if a coalescer pushes more.
|
|
||||||
gsoIovs []unix.Iovec
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// gsoInitialPayIovs is the starting capacity (in payload fragments) of
|
|
||||||
// tunFile.gsoIovs. Sized to cover the default coalesce segment cap without
|
|
||||||
// any reallocations.
|
|
||||||
const gsoInitialPayIovs = 66
|
|
||||||
|
|
||||||
// validVnetHdr is the 10-byte virtio_net_hdr we prepend to every non-GSO TUN
|
|
||||||
// write. Only flag set is VIRTIO_NET_HDR_F_DATA_VALID, which marks the skb
|
|
||||||
// CHECKSUM_UNNECESSARY so the receiving network stack skips L4 checksum
|
|
||||||
// verification. All packets that reach the plain Write / WriteReject paths
|
|
||||||
// already carry a valid L4 checksum (either supplied by a remote peer whose
|
|
||||||
// ciphertext we AEAD-authenticated, or produced by finishChecksum during TSO
|
|
||||||
// segmentation, or built locally by CreateRejectPacket), so trusting them is
|
|
||||||
// safe.
|
|
||||||
var validVnetHdr = [virtioNetHdrLen]byte{unix.VIRTIO_NET_HDR_F_DATA_VALID}
|
|
||||||
|
|
||||||
// newFriend makes a tunFile for a MultiQueueReader that copies the shutdown eventfd from the parent tun
|
// newFriend makes a tunFile for a MultiQueueReader that copies the shutdown eventfd from the parent tun
|
||||||
func (r *tunFile) newFriend(fd int) (*tunFile, error) {
|
func (r *tunFile) newFriend(fd int) (*tunFile, error) {
|
||||||
if err := unix.SetNonblock(fd, true); err != nil {
|
if err := unix.SetNonblock(fd, true); err != nil {
|
||||||
return nil, fmt.Errorf("failed to set tun fd non-blocking: %w", err)
|
return nil, fmt.Errorf("failed to set tun fd non-blocking: %w", err)
|
||||||
}
|
}
|
||||||
out := &tunFile{
|
return &tunFile{
|
||||||
fd: fd,
|
fd: fd,
|
||||||
shutdownFd: r.shutdownFd,
|
shutdownFd: r.shutdownFd,
|
||||||
vnetHdr: r.vnetHdr,
|
|
||||||
readBuf: make([]byte, tunReadBufSize),
|
|
||||||
readPoll: [2]unix.PollFd{
|
readPoll: [2]unix.PollFd{
|
||||||
{Fd: int32(fd), Events: unix.POLLIN},
|
{Fd: int32(fd), Events: unix.POLLIN},
|
||||||
{Fd: int32(r.shutdownFd), Events: unix.POLLIN},
|
{Fd: int32(r.shutdownFd), Events: unix.POLLIN},
|
||||||
@@ -96,21 +52,10 @@ func (r *tunFile) newFriend(fd int) (*tunFile, error) {
|
|||||||
{Fd: int32(fd), Events: unix.POLLOUT},
|
{Fd: int32(fd), Events: unix.POLLOUT},
|
||||||
{Fd: int32(r.shutdownFd), Events: unix.POLLIN},
|
{Fd: int32(r.shutdownFd), Events: unix.POLLIN},
|
||||||
},
|
},
|
||||||
}
|
}, nil
|
||||||
if r.vnetHdr {
|
|
||||||
out.segBuf = make([]byte, tunSegBufCap)
|
|
||||||
out.writeIovs[0].Base = &validVnetHdr[0]
|
|
||||||
out.writeIovs[0].SetLen(virtioNetHdrLen)
|
|
||||||
out.rejectIovs[0].Base = &validVnetHdr[0]
|
|
||||||
out.rejectIovs[0].SetLen(virtioNetHdrLen)
|
|
||||||
out.gsoIovs = make([]unix.Iovec, 2, 2+gsoInitialPayIovs)
|
|
||||||
out.gsoIovs[0].Base = &out.gsoHdrBuf[0]
|
|
||||||
out.gsoIovs[0].SetLen(virtioNetHdrLen)
|
|
||||||
}
|
|
||||||
return out, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func newTunFd(fd int, vnetHdr bool) (*tunFile, error) {
|
func newTunFd(fd int) (*tunFile, error) {
|
||||||
if err := unix.SetNonblock(fd, true); err != nil {
|
if err := unix.SetNonblock(fd, true); err != nil {
|
||||||
return nil, fmt.Errorf("failed to set tun fd non-blocking: %w", err)
|
return nil, fmt.Errorf("failed to set tun fd non-blocking: %w", err)
|
||||||
}
|
}
|
||||||
@@ -124,8 +69,6 @@ func newTunFd(fd int, vnetHdr bool) (*tunFile, error) {
|
|||||||
fd: fd,
|
fd: fd,
|
||||||
shutdownFd: shutdownFd,
|
shutdownFd: shutdownFd,
|
||||||
lastOne: true,
|
lastOne: true,
|
||||||
vnetHdr: vnetHdr,
|
|
||||||
readBuf: make([]byte, tunReadBufSize),
|
|
||||||
readPoll: [2]unix.PollFd{
|
readPoll: [2]unix.PollFd{
|
||||||
{Fd: int32(fd), Events: unix.POLLIN},
|
{Fd: int32(fd), Events: unix.POLLIN},
|
||||||
{Fd: int32(shutdownFd), Events: unix.POLLIN},
|
{Fd: int32(shutdownFd), Events: unix.POLLIN},
|
||||||
@@ -135,16 +78,6 @@ func newTunFd(fd int, vnetHdr bool) (*tunFile, error) {
|
|||||||
{Fd: int32(shutdownFd), Events: unix.POLLIN},
|
{Fd: int32(shutdownFd), Events: unix.POLLIN},
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
if vnetHdr {
|
|
||||||
out.segBuf = make([]byte, tunSegBufCap)
|
|
||||||
out.writeIovs[0].Base = &validVnetHdr[0]
|
|
||||||
out.writeIovs[0].SetLen(virtioNetHdrLen)
|
|
||||||
out.rejectIovs[0].Base = &validVnetHdr[0]
|
|
||||||
out.rejectIovs[0].SetLen(virtioNetHdrLen)
|
|
||||||
out.gsoIovs = make([]unix.Iovec, 2, 2+gsoInitialPayIovs)
|
|
||||||
out.gsoIovs[0].Base = &out.gsoHdrBuf[0]
|
|
||||||
out.gsoIovs[0].SetLen(virtioNetHdrLen)
|
|
||||||
}
|
|
||||||
|
|
||||||
return out, nil
|
return out, nil
|
||||||
}
|
}
|
||||||
@@ -201,7 +134,7 @@ func (r *tunFile) blockOnWrite() error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (r *tunFile) readRaw(buf []byte) (int, error) {
|
func (r *tunFile) Read(buf []byte) (int, error) {
|
||||||
for {
|
for {
|
||||||
if n, err := unix.Read(r.fd, buf); err == nil {
|
if n, err := unix.Read(r.fd, buf); err == nil {
|
||||||
return n, nil
|
return n, nil
|
||||||
@@ -220,238 +153,22 @@ func (r *tunFile) readRaw(buf []byte) (int, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Read reads one or more superpackets from the tun and returns the
|
|
||||||
// resulting packets. The first read blocks via poll; once the fd is known
|
|
||||||
// readable we drain additional packets non-blocking until the kernel queue
|
|
||||||
// is empty (EAGAIN), we've collected tunDrainCap packets, or we're out of
|
|
||||||
// segBuf headroom. This amortizes the poll wake over bursts of small
|
|
||||||
// packets (e.g. TCP ACKs). Slices point into the tunFile's internal buffers
|
|
||||||
// and are only valid until the next Read or Close on this Queue.
|
|
||||||
func (r *tunFile) Read() ([][]byte, error) {
|
|
||||||
r.pending = r.pending[:0]
|
|
||||||
r.segOff = 0
|
|
||||||
|
|
||||||
// Initial (blocking) read. Retry on decode errors so a single bad
|
|
||||||
// packet does not stall the reader.
|
|
||||||
for {
|
|
||||||
n, err := r.readRaw(r.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if !r.vnetHdr {
|
|
||||||
r.pending = append(r.pending, r.readBuf[:n])
|
|
||||||
// Non-vnetHdr mode shares one readBuf so we can't drain safely
|
|
||||||
// without copying; return the single packet as before.
|
|
||||||
return r.pending, nil
|
|
||||||
}
|
|
||||||
if err := r.decodeRead(n); err != nil {
|
|
||||||
// Drop and read again — a bad packet should not kill the reader.
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
break
|
|
||||||
}
|
|
||||||
|
|
||||||
// Drain: non-blocking reads until the kernel queue is empty, the drain
|
|
||||||
// cap is reached, or segBuf no longer has room for another worst-case
|
|
||||||
// superpacket.
|
|
||||||
for len(r.pending) < tunDrainCap && tunSegBufCap-r.segOff >= tunSegBufSize {
|
|
||||||
n, err := unix.Read(r.fd, r.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
// EAGAIN / EINTR / anything else: stop draining. We already
|
|
||||||
// have a valid batch from the first read.
|
|
||||||
break
|
|
||||||
}
|
|
||||||
if n <= 0 {
|
|
||||||
break
|
|
||||||
}
|
|
||||||
if err := r.decodeRead(n); err != nil {
|
|
||||||
// Drop this packet and stop the drain; we'd rather hand off
|
|
||||||
// what we have than keep spinning here.
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return r.pending, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// decodeRead decodes the virtio header plus payload in r.readBuf[:n], appends
|
|
||||||
// the segments to r.pending, and advances r.segOff by the total scratch used.
|
|
||||||
// Caller must have already ensured r.vnetHdr is true.
|
|
||||||
func (r *tunFile) decodeRead(n int) error {
|
|
||||||
if n < virtioNetHdrLen {
|
|
||||||
return fmt.Errorf("short tun read: %d < %d", n, virtioNetHdrLen)
|
|
||||||
}
|
|
||||||
var hdr virtioNetHdr
|
|
||||||
hdr.decode(r.readBuf[:virtioNetHdrLen])
|
|
||||||
before := len(r.pending)
|
|
||||||
if err := segmentInto(r.readBuf[virtioNetHdrLen:n], hdr, &r.pending, r.segBuf[r.segOff:]); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
for k := before; k < len(r.pending); k++ {
|
|
||||||
r.segOff += len(r.pending[k])
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (r *tunFile) Write(buf []byte) (int, error) {
|
func (r *tunFile) Write(buf []byte) (int, error) {
|
||||||
return r.writeWithScratch(buf, &r.writeIovs)
|
|
||||||
}
|
|
||||||
|
|
||||||
// WriteReject emits a packet using a dedicated iovec scratch (rejectIovs)
|
|
||||||
// distinct from the one used by the coalescer's Write path. This avoids a
|
|
||||||
// data race between the inside (listenIn) goroutine emitting reject or
|
|
||||||
// self-forward packets and the outside (listenOut) goroutine flushing TCP
|
|
||||||
// coalescer passthroughs on the same tunFile.
|
|
||||||
func (r *tunFile) WriteReject(buf []byte) (int, error) {
|
|
||||||
return r.writeWithScratch(buf, &r.rejectIovs)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (r *tunFile) writeWithScratch(buf []byte, iovs *[2]unix.Iovec) (int, error) {
|
|
||||||
if !r.vnetHdr {
|
|
||||||
for {
|
|
||||||
if n, err := unix.Write(r.fd, buf); err == nil {
|
|
||||||
return n, nil
|
|
||||||
} else if err == unix.EAGAIN {
|
|
||||||
if err = r.blockOnWrite(); err != nil {
|
|
||||||
return 0, err
|
|
||||||
}
|
|
||||||
continue
|
|
||||||
} else if err == unix.EINTR {
|
|
||||||
continue
|
|
||||||
} else if err == unix.EBADF {
|
|
||||||
return 0, os.ErrClosed
|
|
||||||
} else {
|
|
||||||
return 0, err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(buf) == 0 {
|
|
||||||
return 0, nil
|
|
||||||
}
|
|
||||||
// Point the payload iovec at the caller's buffer. iovs[0] is pre-wired
|
|
||||||
// to validVnetHdr during tunFile construction so we don't rebuild it here.
|
|
||||||
iovs[1].Base = &buf[0]
|
|
||||||
iovs[1].SetLen(len(buf))
|
|
||||||
iovPtr := uintptr(unsafe.Pointer(&iovs[0]))
|
|
||||||
// The TUN fd is non-blocking (set in newTunFd / newFriend), so writev
|
|
||||||
// either completes promptly or returns EAGAIN — it cannot park the
|
|
||||||
// goroutine inside the kernel. That lets us use syscall.RawSyscall and
|
|
||||||
// skip the runtime.entersyscall / exitsyscall bookkeeping on every
|
|
||||||
// packet; we only pay that cost when we fall through to blockOnWrite.
|
|
||||||
for {
|
for {
|
||||||
n, _, errno := syscall.RawSyscall(unix.SYS_WRITEV, uintptr(r.fd), iovPtr, 2)
|
if n, err := unix.Write(r.fd, buf); err == nil {
|
||||||
if errno == 0 {
|
return n, nil
|
||||||
runtime.KeepAlive(buf)
|
} else if err == unix.EAGAIN {
|
||||||
if int(n) < virtioNetHdrLen {
|
if err = r.blockOnWrite(); err != nil {
|
||||||
return 0, io.ErrShortWrite
|
|
||||||
}
|
|
||||||
return int(n) - virtioNetHdrLen, nil
|
|
||||||
}
|
|
||||||
if errno == unix.EAGAIN {
|
|
||||||
runtime.KeepAlive(buf)
|
|
||||||
if err := r.blockOnWrite(); err != nil {
|
|
||||||
return 0, err
|
return 0, err
|
||||||
}
|
}
|
||||||
continue
|
continue
|
||||||
}
|
} else if err == unix.EINTR {
|
||||||
if errno == unix.EINTR {
|
|
||||||
continue
|
continue
|
||||||
}
|
} else if err == unix.EBADF {
|
||||||
runtime.KeepAlive(buf)
|
return 0, os.ErrClosed
|
||||||
return 0, errno
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// GSOSupported reports whether this queue was opened with IFF_VNET_HDR and
|
|
||||||
// can accept WriteGSO. When false, callers should fall back to per-segment
|
|
||||||
// Write calls.
|
|
||||||
func (r *tunFile) GSOSupported() bool { return r.vnetHdr }
|
|
||||||
|
|
||||||
// WriteGSO emits a TCP TSO superpacket in a single writev. hdr is the
|
|
||||||
// IPv4/IPv6 + TCP header prefix (already finalized — total length, IP csum,
|
|
||||||
// and TCP pseudo-header partial set by the caller). pays are payload
|
|
||||||
// fragments whose concatenation forms the full coalesced payload; each
|
|
||||||
// slice is read-only and must stay valid until return. gsoSize is the MSS;
|
|
||||||
// every segment except possibly the last is exactly gsoSize bytes.
|
|
||||||
// csumStart is the byte offset where the TCP header begins within hdr.
|
|
||||||
func (r *tunFile) WriteGSO(hdr []byte, pays [][]byte, gsoSize uint16, isV6 bool, csumStart uint16) error {
|
|
||||||
if !r.vnetHdr {
|
|
||||||
return fmt.Errorf("WriteGSO called on tun without IFF_VNET_HDR")
|
|
||||||
}
|
|
||||||
if len(hdr) == 0 || len(pays) == 0 {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// Build the virtio_net_hdr. When pays total to <= gsoSize the kernel
|
|
||||||
// would produce a single segment; keep NEEDS_CSUM semantics but skip
|
|
||||||
// the GSO type so the kernel doesn't spuriously mark this as TSO.
|
|
||||||
vhdr := virtioNetHdr{
|
|
||||||
Flags: unix.VIRTIO_NET_HDR_F_NEEDS_CSUM,
|
|
||||||
HdrLen: uint16(len(hdr)),
|
|
||||||
GSOSize: gsoSize,
|
|
||||||
CsumStart: csumStart,
|
|
||||||
CsumOffset: 16, // TCP checksum field lives 16 bytes into the TCP header
|
|
||||||
}
|
|
||||||
var totalPay int
|
|
||||||
for _, p := range pays {
|
|
||||||
totalPay += len(p)
|
|
||||||
}
|
|
||||||
if totalPay > int(gsoSize) {
|
|
||||||
if isV6 {
|
|
||||||
vhdr.GSOType = unix.VIRTIO_NET_HDR_GSO_TCPV6
|
|
||||||
} else {
|
} else {
|
||||||
vhdr.GSOType = unix.VIRTIO_NET_HDR_GSO_TCPV4
|
return 0, err
|
||||||
}
|
}
|
||||||
} else {
|
|
||||||
vhdr.GSOType = unix.VIRTIO_NET_HDR_GSO_NONE
|
|
||||||
vhdr.GSOSize = 0
|
|
||||||
}
|
|
||||||
vhdr.encode(r.gsoHdrBuf[:])
|
|
||||||
|
|
||||||
// Build the iovec array: [virtio_hdr, hdr, pays...]. r.gsoIovs[0] is
|
|
||||||
// wired to gsoHdrBuf at construction and never changes.
|
|
||||||
need := 2 + len(pays)
|
|
||||||
if cap(r.gsoIovs) < need {
|
|
||||||
grown := make([]unix.Iovec, need)
|
|
||||||
grown[0] = r.gsoIovs[0]
|
|
||||||
r.gsoIovs = grown
|
|
||||||
} else {
|
|
||||||
r.gsoIovs = r.gsoIovs[:need]
|
|
||||||
}
|
|
||||||
r.gsoIovs[1].Base = &hdr[0]
|
|
||||||
r.gsoIovs[1].SetLen(len(hdr))
|
|
||||||
for i, p := range pays {
|
|
||||||
r.gsoIovs[2+i].Base = &p[0]
|
|
||||||
r.gsoIovs[2+i].SetLen(len(p))
|
|
||||||
}
|
|
||||||
|
|
||||||
iovPtr := uintptr(unsafe.Pointer(&r.gsoIovs[0]))
|
|
||||||
iovCnt := uintptr(len(r.gsoIovs))
|
|
||||||
for {
|
|
||||||
n, _, errno := syscall.RawSyscall(unix.SYS_WRITEV, uintptr(r.fd), iovPtr, iovCnt)
|
|
||||||
if errno == 0 {
|
|
||||||
runtime.KeepAlive(hdr)
|
|
||||||
runtime.KeepAlive(pays)
|
|
||||||
if int(n) < virtioNetHdrLen {
|
|
||||||
return io.ErrShortWrite
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
if errno == unix.EAGAIN {
|
|
||||||
runtime.KeepAlive(hdr)
|
|
||||||
runtime.KeepAlive(pays)
|
|
||||||
if err := r.blockOnWrite(); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if errno == unix.EINTR {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
runtime.KeepAlive(hdr)
|
|
||||||
runtime.KeepAlive(pays)
|
|
||||||
return errno
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -522,9 +239,7 @@ type ifreqQLEN struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []netip.Prefix) (*tun, error) {
|
func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []netip.Prefix) (*tun, error) {
|
||||||
// We don't know what flags the caller opened this fd with and can't turn
|
t, err := newTunGeneric(c, l, deviceFd, vpnNetworks)
|
||||||
// on IFF_VNET_HDR after TUNSETIFF, so skip offload on inherited fds.
|
|
||||||
t, err := newTunGeneric(c, l, deviceFd, false, vpnNetworks)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -534,83 +249,46 @@ func newTunFromFd(c *config.C, l *logrus.Logger, deviceFd int, vpnNetworks []net
|
|||||||
return t, nil
|
return t, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// openTunDev opens /dev/net/tun, creating the device node first if it's
|
|
||||||
// missing (docker containers occasionally omit it).
|
|
||||||
func openTunDev() (int, error) {
|
|
||||||
fd, err := unix.Open("/dev/net/tun", os.O_RDWR, 0)
|
|
||||||
if err == nil {
|
|
||||||
return fd, nil
|
|
||||||
}
|
|
||||||
if !os.IsNotExist(err) {
|
|
||||||
return -1, err
|
|
||||||
}
|
|
||||||
if err = os.MkdirAll("/dev/net", 0755); err != nil {
|
|
||||||
return -1, fmt.Errorf("/dev/net/tun doesn't exist, failed to mkdir -p /dev/net: %w", err)
|
|
||||||
}
|
|
||||||
if err = unix.Mknod("/dev/net/tun", unix.S_IFCHR|0600, int(unix.Mkdev(10, 200))); err != nil {
|
|
||||||
return -1, fmt.Errorf("failed to create /dev/net/tun: %w", err)
|
|
||||||
}
|
|
||||||
fd, err = unix.Open("/dev/net/tun", os.O_RDWR, 0)
|
|
||||||
if err != nil {
|
|
||||||
return -1, fmt.Errorf("created /dev/net/tun, but still failed: %w", err)
|
|
||||||
}
|
|
||||||
return fd, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// tunSetIff runs TUNSETIFF with the given flags and returns the kernel-chosen
|
|
||||||
// device name on success.
|
|
||||||
func tunSetIff(fd int, name string, flags uint16) (string, error) {
|
|
||||||
var req ifReq
|
|
||||||
req.Flags = flags
|
|
||||||
copy(req.Name[:], name)
|
|
||||||
if err := ioctl(uintptr(fd), uintptr(unix.TUNSETIFF), uintptr(unsafe.Pointer(&req))); err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
return strings.Trim(string(req.Name[:]), "\x00"), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// tsoOffloadFlags are the TUN_F_* bits we ask the kernel to enable when a
|
|
||||||
// TSO-capable TUN is available. CSUM is required as a prerequisite for TSO.
|
|
||||||
const tsoOffloadFlags = unix.TUN_F_CSUM | unix.TUN_F_TSO4 | unix.TUN_F_TSO6
|
|
||||||
|
|
||||||
func newTun(c *config.C, l *logrus.Logger, vpnNetworks []netip.Prefix, multiqueue bool) (*tun, error) {
|
func newTun(c *config.C, l *logrus.Logger, vpnNetworks []netip.Prefix, multiqueue bool) (*tun, error) {
|
||||||
baseFlags := uint16(unix.IFF_TUN | unix.IFF_NO_PI)
|
fd, err := unix.Open("/dev/net/tun", os.O_RDWR, 0)
|
||||||
if multiqueue {
|
|
||||||
baseFlags |= unix.IFF_MULTI_QUEUE
|
|
||||||
}
|
|
||||||
nameStr := c.GetString("tun.dev", "")
|
|
||||||
|
|
||||||
// First try to open with IFF_VNET_HDR + TUNSETOFFLOAD so we can receive
|
|
||||||
// TSO superpackets. If either step fails (older kernel, unprivileged
|
|
||||||
// container, etc.) we close and fall back to a plain TUN.
|
|
||||||
fd, err := openTunDev()
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
// If /dev/net/tun doesn't exist, try to create it (will happen in docker)
|
||||||
}
|
if os.IsNotExist(err) {
|
||||||
vnetHdr := true
|
err = os.MkdirAll("/dev/net", 0755)
|
||||||
name, err := tunSetIff(fd, nameStr, baseFlags|unix.IFF_VNET_HDR|unix.IFF_NAPI)
|
if err != nil {
|
||||||
if err != nil {
|
return nil, fmt.Errorf("/dev/net/tun doesn't exist, failed to mkdir -p /dev/net: %w", err)
|
||||||
_ = unix.Close(fd)
|
}
|
||||||
vnetHdr = false
|
err = unix.Mknod("/dev/net/tun", unix.S_IFCHR|0600, int(unix.Mkdev(10, 200)))
|
||||||
} else if err = ioctl(uintptr(fd), unix.TUNSETOFFLOAD, uintptr(tsoOffloadFlags)); err != nil {
|
if err != nil {
|
||||||
l.WithError(err).Warn("Failed to enable TUN offload (TSO); proceeding without virtio headers")
|
return nil, fmt.Errorf("failed to create /dev/net/tun: %w", err)
|
||||||
_ = unix.Close(fd)
|
}
|
||||||
vnetHdr = false
|
|
||||||
}
|
|
||||||
|
|
||||||
if !vnetHdr {
|
fd, err = unix.Open("/dev/net/tun", os.O_RDWR, 0)
|
||||||
fd, err = openTunDev()
|
if err != nil {
|
||||||
if err != nil {
|
return nil, fmt.Errorf("created /dev/net/tun, but still failed: %w", err)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
name, err = tunSetIff(fd, nameStr, baseFlags)
|
|
||||||
if err != nil {
|
|
||||||
_ = unix.Close(fd)
|
|
||||||
return nil, &NameError{Name: nameStr, Underlying: err}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
t, err := newTunGeneric(c, l, fd, vnetHdr, vpnNetworks)
|
var req ifReq
|
||||||
|
req.Flags = uint16(unix.IFF_TUN | unix.IFF_NO_PI)
|
||||||
|
if multiqueue {
|
||||||
|
req.Flags |= unix.IFF_MULTI_QUEUE
|
||||||
|
}
|
||||||
|
nameStr := c.GetString("tun.dev", "")
|
||||||
|
copy(req.Name[:], nameStr)
|
||||||
|
if err = ioctl(uintptr(fd), uintptr(unix.TUNSETIFF), uintptr(unsafe.Pointer(&req))); err != nil {
|
||||||
|
_ = unix.Close(fd)
|
||||||
|
return nil, &NameError{
|
||||||
|
Name: nameStr,
|
||||||
|
Underlying: err,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
name := strings.Trim(string(req.Name[:]), "\x00")
|
||||||
|
|
||||||
|
t, err := newTunGeneric(c, l, fd, vpnNetworks)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -621,8 +299,8 @@ func newTun(c *config.C, l *logrus.Logger, vpnNetworks []netip.Prefix, multiqueu
|
|||||||
}
|
}
|
||||||
|
|
||||||
// newTunGeneric does all the stuff common to different tun initialization paths. It will close your files on error.
|
// newTunGeneric does all the stuff common to different tun initialization paths. It will close your files on error.
|
||||||
func newTunGeneric(c *config.C, l *logrus.Logger, fd int, vnetHdr bool, vpnNetworks []netip.Prefix) (*tun, error) {
|
func newTunGeneric(c *config.C, l *logrus.Logger, fd int, vpnNetworks []netip.Prefix) (*tun, error) {
|
||||||
tfd, err := newTunFd(fd, vnetHdr)
|
tfd, err := newTunFd(fd)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
_ = unix.Close(fd)
|
_ = unix.Close(fd)
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -732,7 +410,7 @@ func (t *tun) SupportsMultiqueue() bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) NewMultiQueueReader() (Queue, error) {
|
func (t *tun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
t.closeLock.Lock()
|
t.closeLock.Lock()
|
||||||
defer t.closeLock.Unlock()
|
defer t.closeLock.Unlock()
|
||||||
|
|
||||||
@@ -741,22 +419,14 @@ func (t *tun) NewMultiQueueReader() (Queue, error) {
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
flags := uint16(unix.IFF_TUN | unix.IFF_NO_PI | unix.IFF_MULTI_QUEUE)
|
var req ifReq
|
||||||
if t.vnetHdr {
|
req.Flags = uint16(unix.IFF_TUN | unix.IFF_NO_PI | unix.IFF_MULTI_QUEUE)
|
||||||
flags |= unix.IFF_VNET_HDR | unix.IFF_NAPI
|
copy(req.Name[:], t.Device)
|
||||||
}
|
if err = ioctl(uintptr(fd), uintptr(unix.TUNSETIFF), uintptr(unsafe.Pointer(&req))); err != nil {
|
||||||
if _, err = tunSetIff(fd, t.Device, flags); err != nil {
|
|
||||||
_ = unix.Close(fd)
|
_ = unix.Close(fd)
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
if t.vnetHdr {
|
|
||||||
if err = ioctl(uintptr(fd), unix.TUNSETOFFLOAD, uintptr(tsoOffloadFlags)); err != nil {
|
|
||||||
_ = unix.Close(fd)
|
|
||||||
return nil, fmt.Errorf("failed to enable offload on multiqueue tun fd: %w", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
out, err := t.tunFile.newFriend(fd)
|
out, err := t.tunFile.newFriend(fd)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
_ = unix.Close(fd)
|
_ = unix.Close(fd)
|
||||||
|
|||||||
@@ -1,331 +0,0 @@
|
|||||||
//go:build linux && !android && !e2e_testing
|
|
||||||
// +build linux,!android,!e2e_testing
|
|
||||||
|
|
||||||
package overlay
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/binary"
|
|
||||||
"fmt"
|
|
||||||
|
|
||||||
"golang.org/x/sys/unix"
|
|
||||||
)
|
|
||||||
|
|
||||||
// Size of the legacy struct virtio_net_hdr that the kernel prepends/expects on
|
|
||||||
// a TUN opened with IFF_VNET_HDR (TUNSETVNETHDRSZ not set).
|
|
||||||
const virtioNetHdrLen = 10
|
|
||||||
|
|
||||||
// Maximum size we accept for a single read from a TUN with IFF_VNET_HDR. A
|
|
||||||
// TSO superpacket can be up to 64KiB of payload plus a single L2/L3/L4 header
|
|
||||||
// prefix plus the virtio header.
|
|
||||||
const tunReadBufSize = 65535
|
|
||||||
|
|
||||||
// Space for segmented output. Worst case is many small segments, each paying
|
|
||||||
// an IP+TCP header. 128KiB comfortably covers the 64KiB payload ceiling.
|
|
||||||
const tunSegBufSize = 131072
|
|
||||||
|
|
||||||
// tunSegBufCap is the total size we allocate for the per-reader segment
|
|
||||||
// buffer. It is sized as one worst-case TSO superpacket (tunSegBufSize) plus
|
|
||||||
// the same again as drain headroom so a Read wake can accumulate
|
|
||||||
// additional packets after an initial big read without overflowing.
|
|
||||||
const tunSegBufCap = tunSegBufSize * 2
|
|
||||||
|
|
||||||
// tunDrainCap caps how many packets a single Read will accumulate via
|
|
||||||
// the post-wake drain loop. Sized to soak up a burst of small ACKs while
|
|
||||||
// bounding how much work a single caller holds before handing off.
|
|
||||||
const tunDrainCap = 64
|
|
||||||
|
|
||||||
type virtioNetHdr struct {
|
|
||||||
Flags uint8
|
|
||||||
GSOType uint8
|
|
||||||
HdrLen uint16
|
|
||||||
GSOSize uint16
|
|
||||||
CsumStart uint16
|
|
||||||
CsumOffset uint16
|
|
||||||
}
|
|
||||||
|
|
||||||
// decode reads a virtio_net_hdr in host byte order (TUN default; we never
|
|
||||||
// call TUNSETVNETLE so the kernel matches our endianness).
|
|
||||||
func (h *virtioNetHdr) decode(b []byte) {
|
|
||||||
h.Flags = b[0]
|
|
||||||
h.GSOType = b[1]
|
|
||||||
h.HdrLen = binary.NativeEndian.Uint16(b[2:4])
|
|
||||||
h.GSOSize = binary.NativeEndian.Uint16(b[4:6])
|
|
||||||
h.CsumStart = binary.NativeEndian.Uint16(b[6:8])
|
|
||||||
h.CsumOffset = binary.NativeEndian.Uint16(b[8:10])
|
|
||||||
}
|
|
||||||
|
|
||||||
// encode is the inverse of decode: writes the virtio_net_hdr fields into b
|
|
||||||
// (must be at least virtioNetHdrLen bytes). Used to emit a TSO superpacket
|
|
||||||
// on egress.
|
|
||||||
func (h *virtioNetHdr) encode(b []byte) {
|
|
||||||
b[0] = h.Flags
|
|
||||||
b[1] = h.GSOType
|
|
||||||
binary.NativeEndian.PutUint16(b[2:4], h.HdrLen)
|
|
||||||
binary.NativeEndian.PutUint16(b[4:6], h.GSOSize)
|
|
||||||
binary.NativeEndian.PutUint16(b[6:8], h.CsumStart)
|
|
||||||
binary.NativeEndian.PutUint16(b[8:10], h.CsumOffset)
|
|
||||||
}
|
|
||||||
|
|
||||||
// segmentInto splits a TUN-side packet described by hdr into one or more
|
|
||||||
// IP packets, each appended to *out as a slice of scratch. scratch must be
|
|
||||||
// sized to hold every segment (including replicated headers).
|
|
||||||
func segmentInto(pkt []byte, hdr virtioNetHdr, out *[][]byte, scratch []byte) error {
|
|
||||||
// When RSC_INFO is set the csum_start/csum_offset fields are repurposed to
|
|
||||||
// carry coalescing info rather than checksum offsets. A TUN writing via
|
|
||||||
// IFF_VNET_HDR should never emit this, but if it did we would silently
|
|
||||||
// miscompute the segment checksums — refuse the packet instead.
|
|
||||||
if hdr.Flags&unix.VIRTIO_NET_HDR_F_RSC_INFO != 0 {
|
|
||||||
return fmt.Errorf("virtio RSC_INFO flag not supported on TUN reads")
|
|
||||||
}
|
|
||||||
|
|
||||||
switch hdr.GSOType {
|
|
||||||
case unix.VIRTIO_NET_HDR_GSO_NONE:
|
|
||||||
if len(pkt) > len(scratch) {
|
|
||||||
return fmt.Errorf("packet larger than segment buffer: %d > %d", len(pkt), len(scratch))
|
|
||||||
}
|
|
||||||
copy(scratch, pkt)
|
|
||||||
seg := scratch[:len(pkt)]
|
|
||||||
if hdr.Flags&unix.VIRTIO_NET_HDR_F_NEEDS_CSUM != 0 {
|
|
||||||
if err := finishChecksum(seg, hdr); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
*out = append(*out, seg)
|
|
||||||
return nil
|
|
||||||
|
|
||||||
case unix.VIRTIO_NET_HDR_GSO_TCPV4, unix.VIRTIO_NET_HDR_GSO_TCPV6:
|
|
||||||
return segmentTCP(pkt, hdr, out, scratch)
|
|
||||||
|
|
||||||
default:
|
|
||||||
return fmt.Errorf("unsupported virtio gso type: %d", hdr.GSOType)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// finishChecksum computes the L4 checksum for a non-GSO packet that the kernel
|
|
||||||
// handed us with NEEDS_CSUM set. csum_start / csum_offset point at the 16-bit
|
|
||||||
// checksum field; we zero it, fold a full sum (the field was pre-loaded with
|
|
||||||
// the pseudo-header partial sum by the kernel), and store the result.
|
|
||||||
func finishChecksum(seg []byte, hdr virtioNetHdr) error {
|
|
||||||
cs := int(hdr.CsumStart)
|
|
||||||
co := int(hdr.CsumOffset)
|
|
||||||
if cs+co+2 > len(seg) {
|
|
||||||
return fmt.Errorf("csum offsets out of range: start=%d offset=%d len=%d", cs, co, len(seg))
|
|
||||||
}
|
|
||||||
// The kernel stores a partial pseudo-header sum at [cs+co:]; sum over the
|
|
||||||
// L4 region starting at cs, folding the prior partial in as the seed.
|
|
||||||
partial := uint32(binary.BigEndian.Uint16(seg[cs+co : cs+co+2]))
|
|
||||||
seg[cs+co] = 0
|
|
||||||
seg[cs+co+1] = 0
|
|
||||||
sum := checksumBytes(seg[cs:], partial)
|
|
||||||
binary.BigEndian.PutUint16(seg[cs+co:cs+co+2], checksumFold(sum))
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// segmentTCP software-segments a TSO superpacket into one IP packet per MSS
|
|
||||||
// chunk. The caller guarantees hdr.GSOType is TCPV4 or TCPV6.
|
|
||||||
//
|
|
||||||
// Hot-path shape: the per-segment loop only sums the payload chunk. The TCP
|
|
||||||
// header, the IPv4 header, and the pseudo-header src/dst/proto contributions
|
|
||||||
// are each summed once up front — every segment reuses those three pre-folded
|
|
||||||
// uint32 values and combines them with small per-segment deltas (seq, flags,
|
|
||||||
// tcpLen, ip_id, total_len) that are cheap to fold in.
|
|
||||||
func segmentTCP(pkt []byte, hdr virtioNetHdr, out *[][]byte, scratch []byte) error {
|
|
||||||
if hdr.GSOSize == 0 {
|
|
||||||
return fmt.Errorf("gso_size is zero")
|
|
||||||
}
|
|
||||||
if int(hdr.HdrLen) > len(pkt) || hdr.HdrLen == 0 {
|
|
||||||
return fmt.Errorf("hdr_len %d out of range (pkt %d)", hdr.HdrLen, len(pkt))
|
|
||||||
}
|
|
||||||
if hdr.CsumStart == 0 || hdr.CsumStart >= hdr.HdrLen {
|
|
||||||
return fmt.Errorf("csum_start %d out of range (hdr_len %d)", hdr.CsumStart, hdr.HdrLen)
|
|
||||||
}
|
|
||||||
|
|
||||||
isV4 := hdr.GSOType == unix.VIRTIO_NET_HDR_GSO_TCPV4
|
|
||||||
headerLen := int(hdr.HdrLen)
|
|
||||||
csumStart := int(hdr.CsumStart)
|
|
||||||
|
|
||||||
if isV4 && csumStart < 20 {
|
|
||||||
return fmt.Errorf("csum_start %d too small for IPv4", csumStart)
|
|
||||||
}
|
|
||||||
if !isV4 && csumStart < 40 {
|
|
||||||
return fmt.Errorf("csum_start %d too small for IPv6", csumStart)
|
|
||||||
}
|
|
||||||
tcpHdrLen := headerLen - csumStart
|
|
||||||
if tcpHdrLen < 20 {
|
|
||||||
return fmt.Errorf("tcp header region too small: %d", tcpHdrLen)
|
|
||||||
}
|
|
||||||
|
|
||||||
payload := pkt[headerLen:]
|
|
||||||
payLen := len(payload)
|
|
||||||
gso := int(hdr.GSOSize)
|
|
||||||
numSeg := (payLen + gso - 1) / gso
|
|
||||||
if numSeg == 0 {
|
|
||||||
numSeg = 1
|
|
||||||
}
|
|
||||||
|
|
||||||
need := numSeg*headerLen + payLen
|
|
||||||
if need > len(scratch) {
|
|
||||||
return fmt.Errorf("scratch too small for %d segments: need %d have %d", numSeg, need, len(scratch))
|
|
||||||
}
|
|
||||||
|
|
||||||
origSeq := binary.BigEndian.Uint32(pkt[csumStart+4 : csumStart+8])
|
|
||||||
origFlags := pkt[csumStart+13]
|
|
||||||
const tcpFinPsh = 0x09 // FIN(0x01) | PSH(0x08)
|
|
||||||
|
|
||||||
// Precompute the TCP header sum with seq/flags/csum zeroed. The max TCP
|
|
||||||
// header is 60 bytes; copy onto the stack, zero the per-segment-varying
|
|
||||||
// fields, sum once.
|
|
||||||
var tmp [60]byte
|
|
||||||
copy(tmp[:tcpHdrLen], pkt[csumStart:headerLen])
|
|
||||||
tmp[4], tmp[5], tmp[6], tmp[7] = 0, 0, 0, 0 // seq
|
|
||||||
tmp[13] = 0 // flags
|
|
||||||
tmp[16], tmp[17] = 0, 0 // csum
|
|
||||||
baseTcpHdrSum := checksumBytes(tmp[:tcpHdrLen], 0)
|
|
||||||
|
|
||||||
// Pseudo-header src+dst+proto contribution (tcpLen varies per segment).
|
|
||||||
var baseProtoSum uint32
|
|
||||||
if isV4 {
|
|
||||||
baseProtoSum = checksumBytes(pkt[12:16], 0)
|
|
||||||
baseProtoSum = checksumBytes(pkt[16:20], baseProtoSum)
|
|
||||||
} else {
|
|
||||||
baseProtoSum = checksumBytes(pkt[8:24], 0)
|
|
||||||
baseProtoSum = checksumBytes(pkt[24:40], baseProtoSum)
|
|
||||||
}
|
|
||||||
baseProtoSum += uint32(unix.IPPROTO_TCP)
|
|
||||||
|
|
||||||
// Precompute IPv4 header sum with total_len/id/csum zeroed.
|
|
||||||
var origIPID uint16
|
|
||||||
var ihl int
|
|
||||||
var baseIPHdrSum uint32
|
|
||||||
if isV4 {
|
|
||||||
origIPID = binary.BigEndian.Uint16(pkt[4:6])
|
|
||||||
ihl = int(pkt[0]&0x0f) * 4
|
|
||||||
if ihl < 20 || ihl > csumStart {
|
|
||||||
return fmt.Errorf("bad IPv4 IHL: %d", ihl)
|
|
||||||
}
|
|
||||||
var ipTmp [60]byte
|
|
||||||
copy(ipTmp[:ihl], pkt[:ihl])
|
|
||||||
ipTmp[2], ipTmp[3] = 0, 0 // total_len
|
|
||||||
ipTmp[4], ipTmp[5] = 0, 0 // id
|
|
||||||
ipTmp[10], ipTmp[11] = 0, 0 // checksum
|
|
||||||
baseIPHdrSum = checksumBytes(ipTmp[:ihl], 0)
|
|
||||||
}
|
|
||||||
|
|
||||||
off := 0
|
|
||||||
for i := 0; i < numSeg; i++ {
|
|
||||||
segStart := i * gso
|
|
||||||
segEnd := segStart + gso
|
|
||||||
if segEnd > payLen {
|
|
||||||
segEnd = payLen
|
|
||||||
}
|
|
||||||
segPayLen := segEnd - segStart
|
|
||||||
|
|
||||||
copy(scratch[off:], pkt[:headerLen])
|
|
||||||
copy(scratch[off+headerLen:], payload[segStart:segEnd])
|
|
||||||
seg := scratch[off : off+headerLen+segPayLen]
|
|
||||||
off += headerLen + segPayLen
|
|
||||||
|
|
||||||
segSeq := origSeq + uint32(segStart)
|
|
||||||
segFlags := origFlags
|
|
||||||
if i != numSeg-1 {
|
|
||||||
segFlags = origFlags &^ tcpFinPsh
|
|
||||||
}
|
|
||||||
totalLen := headerLen + segPayLen
|
|
||||||
|
|
||||||
// Patch IP header and write the v4 header checksum from the precomputed base.
|
|
||||||
if isV4 {
|
|
||||||
segID := origIPID + uint16(i)
|
|
||||||
binary.BigEndian.PutUint16(seg[2:4], uint16(totalLen))
|
|
||||||
binary.BigEndian.PutUint16(seg[4:6], segID)
|
|
||||||
ipSum := baseIPHdrSum + uint32(totalLen) + uint32(segID)
|
|
||||||
binary.BigEndian.PutUint16(seg[10:12], checksumFold(ipSum))
|
|
||||||
} else {
|
|
||||||
// IPv6 payload length excludes the 40-byte fixed header but
|
|
||||||
// includes any extension headers between [40:csumStart].
|
|
||||||
binary.BigEndian.PutUint16(seg[4:6], uint16(headerLen-40+segPayLen))
|
|
||||||
}
|
|
||||||
|
|
||||||
// Patch TCP header.
|
|
||||||
binary.BigEndian.PutUint32(seg[csumStart+4:csumStart+8], segSeq)
|
|
||||||
seg[csumStart+13] = segFlags
|
|
||||||
// (csum is written below; its prior contents in `seg` don't affect the
|
|
||||||
// computation since we never sum over the segment's own header.)
|
|
||||||
|
|
||||||
tcpLen := tcpHdrLen + segPayLen
|
|
||||||
paySum := checksumBytes(payload[segStart:segEnd], 0)
|
|
||||||
|
|
||||||
// Combine pre-folded uint32s into a wider accumulator, then fold. Using
|
|
||||||
// uint64 guards against overflow when segSeq's high bits set.
|
|
||||||
wide := uint64(baseTcpHdrSum) + uint64(paySum) + uint64(baseProtoSum)
|
|
||||||
wide += uint64(segSeq) + uint64(segFlags) + uint64(tcpLen)
|
|
||||||
wide = (wide & 0xffffffff) + (wide >> 32)
|
|
||||||
wide = (wide & 0xffffffff) + (wide >> 32)
|
|
||||||
binary.BigEndian.PutUint16(seg[csumStart+16:csumStart+18], checksumFold(uint32(wide)))
|
|
||||||
|
|
||||||
*out = append(*out, seg)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// checksumBytes returns the Internet-checksum partial sum of b, seeded with
|
|
||||||
// initial. Result is a 32-bit accumulator; the caller folds to 16.
|
|
||||||
//
|
|
||||||
// Each 4-byte load is added directly into a 64-bit accumulator. Two parallel
|
|
||||||
// accumulators break the serial dependency through `sum` and let the CPU
|
|
||||||
// overlap independent adds. The final fold from 64 → 32 → 16 handles the
|
|
||||||
// carries that accumulated across the 32-bit lane boundary.
|
|
||||||
func checksumBytes(b []byte, initial uint32) uint32 {
|
|
||||||
s0 := uint64(initial)
|
|
||||||
var s1 uint64
|
|
||||||
for len(b) >= 32 {
|
|
||||||
s0 += uint64(binary.BigEndian.Uint32(b[0:4]))
|
|
||||||
s1 += uint64(binary.BigEndian.Uint32(b[4:8]))
|
|
||||||
s0 += uint64(binary.BigEndian.Uint32(b[8:12]))
|
|
||||||
s1 += uint64(binary.BigEndian.Uint32(b[12:16]))
|
|
||||||
s0 += uint64(binary.BigEndian.Uint32(b[16:20]))
|
|
||||||
s1 += uint64(binary.BigEndian.Uint32(b[20:24]))
|
|
||||||
s0 += uint64(binary.BigEndian.Uint32(b[24:28]))
|
|
||||||
s1 += uint64(binary.BigEndian.Uint32(b[28:32]))
|
|
||||||
b = b[32:]
|
|
||||||
}
|
|
||||||
sum := s0 + s1
|
|
||||||
for len(b) >= 4 {
|
|
||||||
sum += uint64(binary.BigEndian.Uint32(b[:4]))
|
|
||||||
b = b[4:]
|
|
||||||
}
|
|
||||||
if len(b) >= 2 {
|
|
||||||
sum += uint64(binary.BigEndian.Uint16(b[:2]))
|
|
||||||
b = b[2:]
|
|
||||||
}
|
|
||||||
if len(b) == 1 {
|
|
||||||
sum += uint64(b[0]) << 8
|
|
||||||
}
|
|
||||||
sum = (sum & 0xffffffff) + (sum >> 32)
|
|
||||||
sum = (sum & 0xffffffff) + (sum >> 32)
|
|
||||||
return uint32(sum)
|
|
||||||
}
|
|
||||||
|
|
||||||
func checksumFold(sum uint32) uint16 {
|
|
||||||
for sum>>16 != 0 {
|
|
||||||
sum = (sum & 0xffff) + (sum >> 16)
|
|
||||||
}
|
|
||||||
return ^uint16(sum)
|
|
||||||
}
|
|
||||||
|
|
||||||
func pseudoHeaderIPv4(src, dst []byte, proto byte, tcpLen int) uint32 {
|
|
||||||
sum := checksumBytes(src, 0)
|
|
||||||
sum = checksumBytes(dst, sum)
|
|
||||||
sum += uint32(proto)
|
|
||||||
sum += uint32(tcpLen)
|
|
||||||
return sum
|
|
||||||
}
|
|
||||||
|
|
||||||
func pseudoHeaderIPv6(src, dst []byte, proto byte, tcpLen int) uint32 {
|
|
||||||
sum := checksumBytes(src, 0)
|
|
||||||
sum = checksumBytes(dst, sum)
|
|
||||||
sum += uint32(tcpLen >> 16)
|
|
||||||
sum += uint32(tcpLen & 0xffff)
|
|
||||||
sum += uint32(proto)
|
|
||||||
return sum
|
|
||||||
}
|
|
||||||
@@ -1,333 +0,0 @@
|
|||||||
//go:build linux && !android && !e2e_testing
|
|
||||||
// +build linux,!android,!e2e_testing
|
|
||||||
|
|
||||||
package overlay
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/binary"
|
|
||||||
"os"
|
|
||||||
"testing"
|
|
||||||
|
|
||||||
"golang.org/x/sys/unix"
|
|
||||||
)
|
|
||||||
|
|
||||||
// verifyChecksum confirms that the one's-complement sum across `b`, optionally
|
|
||||||
// seeded with a pseudo-header sum, folds to all-ones (valid).
|
|
||||||
func verifyChecksum(b []byte, pseudo uint32) bool {
|
|
||||||
sum := checksumBytes(b, pseudo)
|
|
||||||
for sum>>16 != 0 {
|
|
||||||
sum = (sum & 0xffff) + (sum >> 16)
|
|
||||||
}
|
|
||||||
return uint16(sum) == 0xffff
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildTSOv4 builds a synthetic IPv4/TCP TSO superpacket with a payload of
|
|
||||||
// `payLen` bytes split at `mss`.
|
|
||||||
func buildTSOv4(t *testing.T, payLen, mss int) ([]byte, virtioNetHdr) {
|
|
||||||
t.Helper()
|
|
||||||
const ipLen = 20
|
|
||||||
const tcpLen = 20
|
|
||||||
pkt := make([]byte, ipLen+tcpLen+payLen)
|
|
||||||
|
|
||||||
// IPv4 header
|
|
||||||
pkt[0] = 0x45 // version 4, IHL 5
|
|
||||||
// total length is meaningless for TSO but set it anyway
|
|
||||||
binary.BigEndian.PutUint16(pkt[2:4], uint16(ipLen+tcpLen+payLen))
|
|
||||||
binary.BigEndian.PutUint16(pkt[4:6], 0x4242) // original ID
|
|
||||||
pkt[8] = 64 // TTL
|
|
||||||
pkt[9] = unix.IPPROTO_TCP
|
|
||||||
copy(pkt[12:16], []byte{10, 0, 0, 1}) // src
|
|
||||||
copy(pkt[16:20], []byte{10, 0, 0, 2}) // dst
|
|
||||||
|
|
||||||
// TCP header
|
|
||||||
binary.BigEndian.PutUint16(pkt[20:22], 12345) // sport
|
|
||||||
binary.BigEndian.PutUint16(pkt[22:24], 80) // dport
|
|
||||||
binary.BigEndian.PutUint32(pkt[24:28], 10000) // seq
|
|
||||||
binary.BigEndian.PutUint32(pkt[28:32], 20000) // ack
|
|
||||||
pkt[32] = 0x50 // data offset 5 words
|
|
||||||
pkt[33] = 0x18 // ACK | PSH
|
|
||||||
binary.BigEndian.PutUint16(pkt[34:36], 65535) // window
|
|
||||||
|
|
||||||
// payload
|
|
||||||
for i := 0; i < payLen; i++ {
|
|
||||||
pkt[ipLen+tcpLen+i] = byte(i & 0xff)
|
|
||||||
}
|
|
||||||
|
|
||||||
return pkt, virtioNetHdr{
|
|
||||||
Flags: unix.VIRTIO_NET_HDR_F_NEEDS_CSUM,
|
|
||||||
GSOType: unix.VIRTIO_NET_HDR_GSO_TCPV4,
|
|
||||||
HdrLen: uint16(ipLen + tcpLen),
|
|
||||||
GSOSize: uint16(mss),
|
|
||||||
CsumStart: uint16(ipLen),
|
|
||||||
CsumOffset: 16,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSegmentTCPv4(t *testing.T) {
|
|
||||||
const mss = 100
|
|
||||||
const numSeg = 3
|
|
||||||
pkt, hdr := buildTSOv4(t, mss*numSeg, mss)
|
|
||||||
|
|
||||||
scratch := make([]byte, tunSegBufSize)
|
|
||||||
var out [][]byte
|
|
||||||
if err := segmentTCP(pkt, hdr, &out, scratch); err != nil {
|
|
||||||
t.Fatalf("segmentTCP: %v", err)
|
|
||||||
}
|
|
||||||
if len(out) != numSeg {
|
|
||||||
t.Fatalf("expected %d segments, got %d", numSeg, len(out))
|
|
||||||
}
|
|
||||||
|
|
||||||
for i, seg := range out {
|
|
||||||
if len(seg) != 40+mss {
|
|
||||||
t.Errorf("seg %d: unexpected len %d", i, len(seg))
|
|
||||||
}
|
|
||||||
totalLen := binary.BigEndian.Uint16(seg[2:4])
|
|
||||||
if totalLen != uint16(40+mss) {
|
|
||||||
t.Errorf("seg %d: total_len=%d want %d", i, totalLen, 40+mss)
|
|
||||||
}
|
|
||||||
id := binary.BigEndian.Uint16(seg[4:6])
|
|
||||||
if id != 0x4242+uint16(i) {
|
|
||||||
t.Errorf("seg %d: ip id=%#x want %#x", i, id, 0x4242+uint16(i))
|
|
||||||
}
|
|
||||||
seq := binary.BigEndian.Uint32(seg[24:28])
|
|
||||||
wantSeq := uint32(10000 + i*mss)
|
|
||||||
if seq != wantSeq {
|
|
||||||
t.Errorf("seg %d: seq=%d want %d", i, seq, wantSeq)
|
|
||||||
}
|
|
||||||
flags := seg[33]
|
|
||||||
wantFlags := byte(0x10) // ACK only, PSH cleared
|
|
||||||
if i == numSeg-1 {
|
|
||||||
wantFlags = 0x18 // ACK | PSH preserved on last
|
|
||||||
}
|
|
||||||
if flags != wantFlags {
|
|
||||||
t.Errorf("seg %d: flags=%#x want %#x", i, flags, wantFlags)
|
|
||||||
}
|
|
||||||
// IPv4 header checksum must verify against itself.
|
|
||||||
if !verifyChecksum(seg[:20], 0) {
|
|
||||||
t.Errorf("seg %d: bad IPv4 header checksum", i)
|
|
||||||
}
|
|
||||||
// TCP checksum must verify against the pseudo-header.
|
|
||||||
psum := pseudoHeaderIPv4(seg[12:16], seg[16:20], unix.IPPROTO_TCP, 20+mss)
|
|
||||||
if !verifyChecksum(seg[20:], psum) {
|
|
||||||
t.Errorf("seg %d: bad TCP checksum", i)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSegmentTCPv4OddTail(t *testing.T) {
|
|
||||||
// Payload of 250 bytes with MSS 100 → segments of 100, 100, 50.
|
|
||||||
pkt, hdr := buildTSOv4(t, 250, 100)
|
|
||||||
scratch := make([]byte, tunSegBufSize)
|
|
||||||
var out [][]byte
|
|
||||||
if err := segmentTCP(pkt, hdr, &out, scratch); err != nil {
|
|
||||||
t.Fatalf("segmentTCP: %v", err)
|
|
||||||
}
|
|
||||||
if len(out) != 3 {
|
|
||||||
t.Fatalf("want 3 segments, got %d", len(out))
|
|
||||||
}
|
|
||||||
wantPayLens := []int{100, 100, 50}
|
|
||||||
for i, seg := range out {
|
|
||||||
if len(seg)-40 != wantPayLens[i] {
|
|
||||||
t.Errorf("seg %d: pay len %d want %d", i, len(seg)-40, wantPayLens[i])
|
|
||||||
}
|
|
||||||
if !verifyChecksum(seg[:20], 0) {
|
|
||||||
t.Errorf("seg %d: bad IPv4 header checksum", i)
|
|
||||||
}
|
|
||||||
psum := pseudoHeaderIPv4(seg[12:16], seg[16:20], unix.IPPROTO_TCP, 20+wantPayLens[i])
|
|
||||||
if !verifyChecksum(seg[20:], psum) {
|
|
||||||
t.Errorf("seg %d: bad TCP checksum", i)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSegmentTCPv6(t *testing.T) {
|
|
||||||
const ipLen = 40
|
|
||||||
const tcpLen = 20
|
|
||||||
const mss = 120
|
|
||||||
const numSeg = 2
|
|
||||||
payLen := mss * numSeg
|
|
||||||
pkt := make([]byte, ipLen+tcpLen+payLen)
|
|
||||||
|
|
||||||
// IPv6 header
|
|
||||||
pkt[0] = 0x60 // version 6
|
|
||||||
binary.BigEndian.PutUint16(pkt[4:6], uint16(tcpLen+payLen))
|
|
||||||
pkt[6] = unix.IPPROTO_TCP
|
|
||||||
pkt[7] = 64
|
|
||||||
// src/dst fe80::1 / fe80::2
|
|
||||||
pkt[8] = 0xfe
|
|
||||||
pkt[9] = 0x80
|
|
||||||
pkt[23] = 1
|
|
||||||
pkt[24] = 0xfe
|
|
||||||
pkt[25] = 0x80
|
|
||||||
pkt[39] = 2
|
|
||||||
|
|
||||||
// TCP header
|
|
||||||
binary.BigEndian.PutUint16(pkt[40:42], 12345)
|
|
||||||
binary.BigEndian.PutUint16(pkt[42:44], 80)
|
|
||||||
binary.BigEndian.PutUint32(pkt[44:48], 7)
|
|
||||||
binary.BigEndian.PutUint32(pkt[48:52], 99)
|
|
||||||
pkt[52] = 0x50
|
|
||||||
pkt[53] = 0x19 // FIN | ACK | PSH — exercise FIN clearing too
|
|
||||||
binary.BigEndian.PutUint16(pkt[54:56], 65535)
|
|
||||||
|
|
||||||
for i := 0; i < payLen; i++ {
|
|
||||||
pkt[ipLen+tcpLen+i] = byte(i)
|
|
||||||
}
|
|
||||||
|
|
||||||
hdr := virtioNetHdr{
|
|
||||||
Flags: unix.VIRTIO_NET_HDR_F_NEEDS_CSUM,
|
|
||||||
GSOType: unix.VIRTIO_NET_HDR_GSO_TCPV6,
|
|
||||||
HdrLen: uint16(ipLen + tcpLen),
|
|
||||||
GSOSize: uint16(mss),
|
|
||||||
CsumStart: uint16(ipLen),
|
|
||||||
CsumOffset: 16,
|
|
||||||
}
|
|
||||||
|
|
||||||
scratch := make([]byte, tunSegBufSize)
|
|
||||||
var out [][]byte
|
|
||||||
if err := segmentTCP(pkt, hdr, &out, scratch); err != nil {
|
|
||||||
t.Fatalf("segmentTCP: %v", err)
|
|
||||||
}
|
|
||||||
if len(out) != numSeg {
|
|
||||||
t.Fatalf("want %d segments, got %d", numSeg, len(out))
|
|
||||||
}
|
|
||||||
|
|
||||||
for i, seg := range out {
|
|
||||||
if len(seg) != ipLen+tcpLen+mss {
|
|
||||||
t.Errorf("seg %d: len %d want %d", i, len(seg), ipLen+tcpLen+mss)
|
|
||||||
}
|
|
||||||
pl := binary.BigEndian.Uint16(seg[4:6])
|
|
||||||
if pl != uint16(tcpLen+mss) {
|
|
||||||
t.Errorf("seg %d: payload_length=%d want %d", i, pl, tcpLen+mss)
|
|
||||||
}
|
|
||||||
seq := binary.BigEndian.Uint32(seg[44:48])
|
|
||||||
if seq != uint32(7+i*mss) {
|
|
||||||
t.Errorf("seg %d: seq=%d want %d", i, seq, 7+i*mss)
|
|
||||||
}
|
|
||||||
flags := seg[53]
|
|
||||||
// Original flags = 0x19 (FIN|ACK|PSH). FIN(0x01)+PSH(0x08) should be
|
|
||||||
// cleared on all but the last; ACK(0x10) always preserved.
|
|
||||||
wantFlags := byte(0x10)
|
|
||||||
if i == numSeg-1 {
|
|
||||||
wantFlags = 0x19
|
|
||||||
}
|
|
||||||
if flags != wantFlags {
|
|
||||||
t.Errorf("seg %d: flags=%#x want %#x", i, flags, wantFlags)
|
|
||||||
}
|
|
||||||
psum := pseudoHeaderIPv6(seg[8:24], seg[24:40], unix.IPPROTO_TCP, tcpLen+mss)
|
|
||||||
if !verifyChecksum(seg[ipLen:], psum) {
|
|
||||||
t.Errorf("seg %d: bad TCP checksum", i)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSegmentGSONonePassesThrough(t *testing.T) {
|
|
||||||
pkt, hdr := buildTSOv4(t, 100, 100)
|
|
||||||
hdr.GSOType = unix.VIRTIO_NET_HDR_GSO_NONE
|
|
||||||
hdr.Flags = 0 // no NEEDS_CSUM, leave packet untouched
|
|
||||||
|
|
||||||
scratch := make([]byte, tunSegBufSize)
|
|
||||||
var out [][]byte
|
|
||||||
if err := segmentInto(pkt, hdr, &out, scratch); err != nil {
|
|
||||||
t.Fatalf("segmentInto: %v", err)
|
|
||||||
}
|
|
||||||
if len(out) != 1 {
|
|
||||||
t.Fatalf("want 1 segment, got %d", len(out))
|
|
||||||
}
|
|
||||||
if len(out[0]) != len(pkt) {
|
|
||||||
t.Fatalf("unexpected length: %d vs %d", len(out[0]), len(pkt))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSegmentRejectsUDP(t *testing.T) {
|
|
||||||
hdr := virtioNetHdr{GSOType: unix.VIRTIO_NET_HDR_GSO_UDP}
|
|
||||||
var out [][]byte
|
|
||||||
if err := segmentInto(nil, hdr, &out, nil); err == nil {
|
|
||||||
t.Fatalf("expected rejection for UDP GSO")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func BenchmarkSegmentTCPv4(b *testing.B) {
|
|
||||||
sizes := []struct {
|
|
||||||
name string
|
|
||||||
payLen int
|
|
||||||
mss int
|
|
||||||
}{
|
|
||||||
{"64KiB_MSS1460", 65000, 1460},
|
|
||||||
{"16KiB_MSS1460", 16384, 1460},
|
|
||||||
{"4KiB_MSS1460", 4096, 1460},
|
|
||||||
}
|
|
||||||
for _, sz := range sizes {
|
|
||||||
b.Run(sz.name, func(b *testing.B) {
|
|
||||||
const ipLen = 20
|
|
||||||
const tcpLen = 20
|
|
||||||
pkt := make([]byte, ipLen+tcpLen+sz.payLen)
|
|
||||||
pkt[0] = 0x45
|
|
||||||
binary.BigEndian.PutUint16(pkt[2:4], uint16(ipLen+tcpLen+sz.payLen))
|
|
||||||
binary.BigEndian.PutUint16(pkt[4:6], 0x4242)
|
|
||||||
pkt[8] = 64
|
|
||||||
pkt[9] = unix.IPPROTO_TCP
|
|
||||||
copy(pkt[12:16], []byte{10, 0, 0, 1})
|
|
||||||
copy(pkt[16:20], []byte{10, 0, 0, 2})
|
|
||||||
binary.BigEndian.PutUint16(pkt[20:22], 12345)
|
|
||||||
binary.BigEndian.PutUint16(pkt[22:24], 80)
|
|
||||||
binary.BigEndian.PutUint32(pkt[24:28], 10000)
|
|
||||||
binary.BigEndian.PutUint32(pkt[28:32], 20000)
|
|
||||||
pkt[32] = 0x50
|
|
||||||
pkt[33] = 0x18
|
|
||||||
binary.BigEndian.PutUint16(pkt[34:36], 65535)
|
|
||||||
for i := 0; i < sz.payLen; i++ {
|
|
||||||
pkt[ipLen+tcpLen+i] = byte(i)
|
|
||||||
}
|
|
||||||
hdr := virtioNetHdr{
|
|
||||||
Flags: unix.VIRTIO_NET_HDR_F_NEEDS_CSUM,
|
|
||||||
GSOType: unix.VIRTIO_NET_HDR_GSO_TCPV4,
|
|
||||||
HdrLen: uint16(ipLen + tcpLen),
|
|
||||||
GSOSize: uint16(sz.mss),
|
|
||||||
CsumStart: uint16(ipLen),
|
|
||||||
CsumOffset: 16,
|
|
||||||
}
|
|
||||||
|
|
||||||
scratch := make([]byte, tunSegBufSize)
|
|
||||||
out := make([][]byte, 0, 64)
|
|
||||||
|
|
||||||
b.SetBytes(int64(len(pkt)))
|
|
||||||
b.ResetTimer()
|
|
||||||
for i := 0; i < b.N; i++ {
|
|
||||||
out = out[:0]
|
|
||||||
if err := segmentTCP(pkt, hdr, &out, scratch); err != nil {
|
|
||||||
b.Fatal(err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestTunFileWriteVnetHdrNoAlloc verifies the IFF_VNET_HDR fast-path write is
|
|
||||||
// allocation-free. We write to /dev/null so every call succeeds synchronously.
|
|
||||||
func TestTunFileWriteVnetHdrNoAlloc(t *testing.T) {
|
|
||||||
fd, err := unix.Open("/dev/null", os.O_WRONLY, 0)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("open /dev/null: %v", err)
|
|
||||||
}
|
|
||||||
t.Cleanup(func() { _ = unix.Close(fd) })
|
|
||||||
|
|
||||||
tf := &tunFile{fd: fd, vnetHdr: true}
|
|
||||||
tf.writeIovs[0].Base = &validVnetHdr[0]
|
|
||||||
tf.writeIovs[0].SetLen(virtioNetHdrLen)
|
|
||||||
|
|
||||||
payload := make([]byte, 1400)
|
|
||||||
// Warm up (first call may trigger one-time internal allocations elsewhere).
|
|
||||||
if _, err := tf.Write(payload); err != nil {
|
|
||||||
t.Fatalf("Write: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
allocs := testing.AllocsPerRun(1000, func() {
|
|
||||||
if _, err := tf.Write(payload); err != nil {
|
|
||||||
t.Fatalf("Write: %v", err)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
if allocs != 0 {
|
|
||||||
t.Fatalf("Write allocated %.1f times per call, want 0", allocs)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
+3
-21
@@ -6,6 +6,7 @@ package overlay
|
|||||||
import (
|
import (
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"os"
|
"os"
|
||||||
"regexp"
|
"regexp"
|
||||||
@@ -65,25 +66,6 @@ type tun struct {
|
|||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
f *os.File
|
f *os.File
|
||||||
fd int
|
fd int
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Read() ([][]byte, error) {
|
|
||||||
if t.readBuf == nil {
|
|
||||||
t.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := t.readOne(t.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t.batchRet[0] = t.readBuf[:n]
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) WriteReject(p []byte) (int, error) {
|
|
||||||
return t.Write(p)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
var deviceNameRE = regexp.MustCompile(`^tun[0-9]+$`)
|
var deviceNameRE = regexp.MustCompile(`^tun[0-9]+$`)
|
||||||
@@ -159,7 +141,7 @@ func (t *tun) Close() error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) readOne(to []byte) (int, error) {
|
func (t *tun) Read(to []byte) (int, error) {
|
||||||
rc, err := t.f.SyscallConn()
|
rc, err := t.f.SyscallConn()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, fmt.Errorf("failed to get syscall conn for tun: %w", err)
|
return 0, fmt.Errorf("failed to get syscall conn for tun: %w", err)
|
||||||
@@ -412,7 +394,7 @@ func (t *tun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) NewMultiQueueReader() (Queue, error) {
|
func (t *tun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented for netbsd")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented for netbsd")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+3
-21
@@ -6,6 +6,7 @@ package overlay
|
|||||||
import (
|
import (
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"os"
|
"os"
|
||||||
"regexp"
|
"regexp"
|
||||||
@@ -58,25 +59,6 @@ type tun struct {
|
|||||||
fd int
|
fd int
|
||||||
// cache out buffer since we need to prepend 4 bytes for tun metadata
|
// cache out buffer since we need to prepend 4 bytes for tun metadata
|
||||||
out []byte
|
out []byte
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) Read() ([][]byte, error) {
|
|
||||||
if t.readBuf == nil {
|
|
||||||
t.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := t.readOne(t.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t.batchRet[0] = t.readBuf[:n]
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *tun) WriteReject(p []byte) (int, error) {
|
|
||||||
return t.Write(p)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
var deviceNameRE = regexp.MustCompile(`^tun[0-9]+$`)
|
var deviceNameRE = regexp.MustCompile(`^tun[0-9]+$`)
|
||||||
@@ -142,7 +124,7 @@ func (t *tun) Close() error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) readOne(to []byte) (int, error) {
|
func (t *tun) Read(to []byte) (int, error) {
|
||||||
buf := make([]byte, len(to)+4)
|
buf := make([]byte, len(to)+4)
|
||||||
|
|
||||||
n, err := t.f.Read(buf)
|
n, err := t.f.Read(buf)
|
||||||
@@ -332,7 +314,7 @@ func (t *tun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *tun) NewMultiQueueReader() (Queue, error) {
|
func (t *tun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented for openbsd")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented for openbsd")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+10
-16
@@ -26,17 +26,6 @@ type TestTun struct {
|
|||||||
closed atomic.Bool
|
closed atomic.Bool
|
||||||
rxPackets chan []byte // Packets to receive into nebula
|
rxPackets chan []byte // Packets to receive into nebula
|
||||||
TxPackets chan []byte // Packets transmitted outside by nebula
|
TxPackets chan []byte // Packets transmitted outside by nebula
|
||||||
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *TestTun) Read() ([][]byte, error) {
|
|
||||||
p, ok := <-t.rxPackets
|
|
||||||
if !ok {
|
|
||||||
return nil, os.ErrClosed
|
|
||||||
}
|
|
||||||
t.batchRet[0] = p
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func newTun(c *config.C, l *logrus.Logger, vpnNetworks []netip.Prefix, _ bool) (*TestTun, error) {
|
func newTun(c *config.C, l *logrus.Logger, vpnNetworks []netip.Prefix, _ bool) (*TestTun, error) {
|
||||||
@@ -126,10 +115,6 @@ func (t *TestTun) Write(b []byte) (n int, err error) {
|
|||||||
return len(b), nil
|
return len(b), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *TestTun) WriteReject(b []byte) (int, error) {
|
|
||||||
return t.Write(b)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *TestTun) Close() error {
|
func (t *TestTun) Close() error {
|
||||||
if t.closed.CompareAndSwap(false, true) {
|
if t.closed.CompareAndSwap(false, true) {
|
||||||
close(t.rxPackets)
|
close(t.rxPackets)
|
||||||
@@ -138,10 +123,19 @@ func (t *TestTun) Close() error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (t *TestTun) Read(b []byte) (int, error) {
|
||||||
|
p, ok := <-t.rxPackets
|
||||||
|
if !ok {
|
||||||
|
return 0, os.ErrClosed
|
||||||
|
}
|
||||||
|
copy(b, p)
|
||||||
|
return len(p), nil
|
||||||
|
}
|
||||||
|
|
||||||
func (t *TestTun) SupportsMultiqueue() bool {
|
func (t *TestTun) SupportsMultiqueue() bool {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *TestTun) NewMultiQueueReader() (Queue, error) {
|
func (t *TestTun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented")
|
||||||
}
|
}
|
||||||
|
|||||||
+6
-20
@@ -6,6 +6,7 @@ package overlay
|
|||||||
import (
|
import (
|
||||||
"crypto"
|
"crypto"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
@@ -35,25 +36,6 @@ type winTun struct {
|
|||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
|
|
||||||
tun *wintun.NativeTun
|
tun *wintun.NativeTun
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *winTun) Read() ([][]byte, error) {
|
|
||||||
if t.readBuf == nil {
|
|
||||||
t.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := t.tun.Read(t.readBuf, 0)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t.batchRet[0] = t.readBuf[:n]
|
|
||||||
return t.batchRet[:], nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *winTun) WriteReject(p []byte) (int, error) {
|
|
||||||
return t.Write(p)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func newTunFromFd(_ *config.C, _ *logrus.Logger, _ int, _ []netip.Prefix) (Device, error) {
|
func newTunFromFd(_ *config.C, _ *logrus.Logger, _ int, _ []netip.Prefix) (Device, error) {
|
||||||
@@ -247,6 +229,10 @@ func (t *winTun) Name() string {
|
|||||||
return t.Device
|
return t.Device
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (t *winTun) Read(b []byte) (int, error) {
|
||||||
|
return t.tun.Read(b, 0)
|
||||||
|
}
|
||||||
|
|
||||||
func (t *winTun) Write(b []byte) (int, error) {
|
func (t *winTun) Write(b []byte) (int, error) {
|
||||||
return t.tun.Write(b, 0)
|
return t.tun.Write(b, 0)
|
||||||
}
|
}
|
||||||
@@ -255,7 +241,7 @@ func (t *winTun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *winTun) NewMultiQueueReader() (Queue, error) {
|
func (t *winTun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, fmt.Errorf("TODO: multiqueue not implemented for windows")
|
return nil, fmt.Errorf("TODO: multiqueue not implemented for windows")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+4
-19
@@ -34,21 +34,6 @@ type UserDevice struct {
|
|||||||
|
|
||||||
inboundReader *io.PipeReader
|
inboundReader *io.PipeReader
|
||||||
inboundWriter *io.PipeWriter
|
inboundWriter *io.PipeWriter
|
||||||
|
|
||||||
readBuf []byte
|
|
||||||
batchRet [1][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
func (d *UserDevice) Read() ([][]byte, error) {
|
|
||||||
if d.readBuf == nil {
|
|
||||||
d.readBuf = make([]byte, defaultBatchBufSize)
|
|
||||||
}
|
|
||||||
n, err := d.outboundReader.Read(d.readBuf)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
d.batchRet[0] = d.readBuf[:n]
|
|
||||||
return d.batchRet[:], nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func (d *UserDevice) Activate() error {
|
func (d *UserDevice) Activate() error {
|
||||||
@@ -65,7 +50,7 @@ func (d *UserDevice) SupportsMultiqueue() bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func (d *UserDevice) NewMultiQueueReader() (Queue, error) {
|
func (d *UserDevice) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return d, nil
|
return d, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -73,12 +58,12 @@ func (d *UserDevice) Pipe() (*io.PipeReader, *io.PipeWriter) {
|
|||||||
return d.inboundReader, d.outboundWriter
|
return d.inboundReader, d.outboundWriter
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (d *UserDevice) Read(p []byte) (n int, err error) {
|
||||||
|
return d.outboundReader.Read(p)
|
||||||
|
}
|
||||||
func (d *UserDevice) Write(p []byte) (n int, err error) {
|
func (d *UserDevice) Write(p []byte) (n int, err error) {
|
||||||
return d.inboundWriter.Write(p)
|
return d.inboundWriter.Write(p)
|
||||||
}
|
}
|
||||||
func (d *UserDevice) WriteReject(p []byte) (n int, err error) {
|
|
||||||
return d.Write(p)
|
|
||||||
}
|
|
||||||
func (d *UserDevice) Close() error {
|
func (d *UserDevice) Close() error {
|
||||||
d.inboundWriter.Close()
|
d.inboundWriter.Close()
|
||||||
d.outboundWriter.Close()
|
d.outboundWriter.Close()
|
||||||
|
|||||||
-484
@@ -1,484 +0,0 @@
|
|||||||
package nebula
|
|
||||||
|
|
||||||
import (
|
|
||||||
"bytes"
|
|
||||||
"encoding/binary"
|
|
||||||
"io"
|
|
||||||
|
|
||||||
"github.com/slackhq/nebula/overlay"
|
|
||||||
)
|
|
||||||
|
|
||||||
// ipProtoTCP is the IANA protocol number for TCP. Hardcoded instead of
|
|
||||||
// reaching for golang.org/x/sys/unix — that package doesn't define the
|
|
||||||
// constant on Windows, which would break cross-compiles even though this
|
|
||||||
// file runs unchanged on every platform.
|
|
||||||
const ipProtoTCP = 6
|
|
||||||
|
|
||||||
// tcpCoalesceBufSize caps total bytes per superpacket. Mirrors the kernel's
|
|
||||||
// sk_gso_max_size of ~64KiB; anything beyond this would be rejected anyway.
|
|
||||||
const tcpCoalesceBufSize = 65535
|
|
||||||
|
|
||||||
// tcpCoalesceMaxSegs caps how many segments we'll coalesce into a single
|
|
||||||
// superpacket. Keeping this well below the kernel's TSO ceiling bounds
|
|
||||||
// latency.
|
|
||||||
const tcpCoalesceMaxSegs = 64
|
|
||||||
|
|
||||||
// tcpCoalesceHdrCap is the scratch space we copy a seed's IP+TCP header
|
|
||||||
// into. IPv6 (40) + TCP with full options (60) = 100 bytes.
|
|
||||||
const tcpCoalesceHdrCap = 100
|
|
||||||
|
|
||||||
// initialSlots is the starting capacity of the slot pool. One flow per
|
|
||||||
// packet is the worst case so this matches a typical UDP recvmmsg batch.
|
|
||||||
const initialSlots = 64
|
|
||||||
|
|
||||||
// flowKey identifies a TCP flow by {src, dst, sport, dport, family}.
|
|
||||||
// Comparable, so linear scans over the slot list stay tight.
|
|
||||||
type flowKey struct {
|
|
||||||
src, dst [16]byte
|
|
||||||
sport, dport uint16
|
|
||||||
isV6 bool
|
|
||||||
}
|
|
||||||
|
|
||||||
// coalesceSlot is one entry in the coalescer's ordered event queue. When
|
|
||||||
// passthrough is true the slot holds a single borrowed packet that must be
|
|
||||||
// emitted verbatim (non-TCP, non-admissible TCP, or oversize seed). When
|
|
||||||
// passthrough is false the slot is an in-progress coalesced superpacket:
|
|
||||||
// hdrBuf is a mutable copy of the seed's IP+TCP header (we patch total
|
|
||||||
// length and pseudo-header partial at flush), and payIovs are *borrowed*
|
|
||||||
// slices from the caller's plaintext buffers — no payload is ever copied.
|
|
||||||
// The caller (listenOut) must keep those buffers alive until Flush.
|
|
||||||
type coalesceSlot struct {
|
|
||||||
passthrough bool
|
|
||||||
rawPkt []byte // borrowed when passthrough
|
|
||||||
|
|
||||||
fk flowKey
|
|
||||||
hdrBuf [tcpCoalesceHdrCap]byte
|
|
||||||
hdrLen int
|
|
||||||
ipHdrLen int
|
|
||||||
isV6 bool
|
|
||||||
gsoSize int
|
|
||||||
numSeg int
|
|
||||||
totalPay int
|
|
||||||
nextSeq uint32
|
|
||||||
// psh closes the chain: set when the last-accepted segment had PSH or
|
|
||||||
// was sub-gsoSize. No further appends after that.
|
|
||||||
psh bool
|
|
||||||
payIovs [][]byte
|
|
||||||
}
|
|
||||||
|
|
||||||
// tcpCoalescer accumulates adjacent in-flow TCP data segments across
|
|
||||||
// multiple concurrent flows and emits each flow's run as a single TSO
|
|
||||||
// superpacket via overlay.GSOWriter. All output — coalesced or not — is
|
|
||||||
// deferred until Flush so arrival order is preserved on the wire. Owns
|
|
||||||
// no locks; one coalescer per TUN write queue.
|
|
||||||
type tcpCoalescer struct {
|
|
||||||
plainW io.Writer
|
|
||||||
gsoW overlay.GSOWriter // nil when the queue doesn't support TSO
|
|
||||||
|
|
||||||
// slots is the ordered event queue. Flush walks it once and emits each
|
|
||||||
// entry as either a WriteGSO (coalesced) or a plainW.Write (passthrough).
|
|
||||||
slots []*coalesceSlot
|
|
||||||
// openSlots maps a flow key to its most recent non-sealed slot, so new
|
|
||||||
// segments can extend an in-progress superpacket in O(1). Slots are
|
|
||||||
// removed from this map when they close (PSH or short-last-segment),
|
|
||||||
// when a non-admissible packet for that flow arrives, or in Flush.
|
|
||||||
openSlots map[flowKey]*coalesceSlot
|
|
||||||
pool []*coalesceSlot // free list for reuse
|
|
||||||
}
|
|
||||||
|
|
||||||
func newTCPCoalescer(w io.Writer) *tcpCoalescer {
|
|
||||||
c := &tcpCoalescer{
|
|
||||||
plainW: w,
|
|
||||||
slots: make([]*coalesceSlot, 0, initialSlots),
|
|
||||||
openSlots: make(map[flowKey]*coalesceSlot, initialSlots),
|
|
||||||
pool: make([]*coalesceSlot, 0, initialSlots),
|
|
||||||
}
|
|
||||||
if gw, ok := w.(overlay.GSOWriter); ok && gw.GSOSupported() {
|
|
||||||
c.gsoW = gw
|
|
||||||
}
|
|
||||||
return c
|
|
||||||
}
|
|
||||||
|
|
||||||
// parsedTCP holds the fields extracted from a single parse so later steps
|
|
||||||
// (admission, slot lookup, canAppend) don't re-walk the header.
|
|
||||||
type parsedTCP struct {
|
|
||||||
fk flowKey
|
|
||||||
ipHdrLen int
|
|
||||||
tcpHdrLen int
|
|
||||||
hdrLen int
|
|
||||||
payLen int
|
|
||||||
seq uint32
|
|
||||||
flags byte
|
|
||||||
}
|
|
||||||
|
|
||||||
// parseTCPBase extracts the flow key and IP/TCP offsets for any TCP packet,
|
|
||||||
// regardless of whether it's admissible for coalescing. Returns ok=false
|
|
||||||
// for non-TCP or malformed input. Accepts IPv4 (no options, no fragmentation)
|
|
||||||
// and IPv6 (no extension headers).
|
|
||||||
func parseTCPBase(pkt []byte) (parsedTCP, bool) {
|
|
||||||
var p parsedTCP
|
|
||||||
if len(pkt) < 20 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
v := pkt[0] >> 4
|
|
||||||
switch v {
|
|
||||||
case 4:
|
|
||||||
ihl := int(pkt[0]&0x0f) * 4
|
|
||||||
if ihl != 20 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
if pkt[9] != ipProtoTCP {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
// Reject actual fragmentation (MF or non-zero frag offset).
|
|
||||||
if binary.BigEndian.Uint16(pkt[6:8])&0x3fff != 0 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
totalLen := int(binary.BigEndian.Uint16(pkt[2:4]))
|
|
||||||
if totalLen > len(pkt) || totalLen < ihl {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
p.ipHdrLen = 20
|
|
||||||
p.fk.isV6 = false
|
|
||||||
copy(p.fk.src[:4], pkt[12:16])
|
|
||||||
copy(p.fk.dst[:4], pkt[16:20])
|
|
||||||
pkt = pkt[:totalLen]
|
|
||||||
case 6:
|
|
||||||
if len(pkt) < 40 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
if pkt[6] != ipProtoTCP {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
payloadLen := int(binary.BigEndian.Uint16(pkt[4:6]))
|
|
||||||
if 40+payloadLen > len(pkt) {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
p.ipHdrLen = 40
|
|
||||||
p.fk.isV6 = true
|
|
||||||
copy(p.fk.src[:], pkt[8:24])
|
|
||||||
copy(p.fk.dst[:], pkt[24:40])
|
|
||||||
pkt = pkt[:40+payloadLen]
|
|
||||||
default:
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(pkt) < p.ipHdrLen+20 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
tcpOff := int(pkt[p.ipHdrLen+12]>>4) * 4
|
|
||||||
if tcpOff < 20 || tcpOff > 60 {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
if len(pkt) < p.ipHdrLen+tcpOff {
|
|
||||||
return p, false
|
|
||||||
}
|
|
||||||
p.tcpHdrLen = tcpOff
|
|
||||||
p.hdrLen = p.ipHdrLen + tcpOff
|
|
||||||
p.payLen = len(pkt) - p.hdrLen
|
|
||||||
p.seq = binary.BigEndian.Uint32(pkt[p.ipHdrLen+4 : p.ipHdrLen+8])
|
|
||||||
p.flags = pkt[p.ipHdrLen+13]
|
|
||||||
p.fk.sport = binary.BigEndian.Uint16(pkt[p.ipHdrLen : p.ipHdrLen+2])
|
|
||||||
p.fk.dport = binary.BigEndian.Uint16(pkt[p.ipHdrLen+2 : p.ipHdrLen+4])
|
|
||||||
return p, true
|
|
||||||
}
|
|
||||||
|
|
||||||
// coalesceable reports whether a parsed TCP segment is eligible for
|
|
||||||
// coalescing. Accepts only ACK or ACK|PSH with a non-empty payload.
|
|
||||||
func (p parsedTCP) coalesceable() bool {
|
|
||||||
const ack = 0x10
|
|
||||||
const psh = 0x08
|
|
||||||
if p.flags&^(ack|psh) != 0 || p.flags&ack == 0 {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
return p.payLen > 0
|
|
||||||
}
|
|
||||||
|
|
||||||
// Add borrows pkt. The caller must keep pkt valid until the next Flush,
|
|
||||||
// whether or not the packet was coalesced — passthrough (non-admissible)
|
|
||||||
// packets are queued and written at Flush time, not synchronously.
|
|
||||||
func (c *tcpCoalescer) Add(pkt []byte) error {
|
|
||||||
if c.gsoW == nil {
|
|
||||||
c.addPassthrough(pkt)
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
info, ok := parseTCPBase(pkt)
|
|
||||||
if !ok {
|
|
||||||
// Non-TCP or malformed — can't possibly collide with an open flow.
|
|
||||||
c.addPassthrough(pkt)
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
if !info.coalesceable() {
|
|
||||||
// TCP but not admissible (SYN/FIN/RST/URG/CWR/ECE or zero-payload).
|
|
||||||
// Seal this flow's open slot so later in-flow packets don't extend
|
|
||||||
// it and accidentally reorder past this passthrough.
|
|
||||||
delete(c.openSlots, info.fk)
|
|
||||||
c.addPassthrough(pkt)
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
if open := c.openSlots[info.fk]; open != nil {
|
|
||||||
if c.canAppend(open, pkt, info) {
|
|
||||||
c.appendPayload(open, pkt, info)
|
|
||||||
if open.psh {
|
|
||||||
delete(c.openSlots, info.fk)
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
// Can't extend — seal it and fall through to seed a fresh slot.
|
|
||||||
delete(c.openSlots, info.fk)
|
|
||||||
}
|
|
||||||
c.seed(pkt, info)
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// Flush emits every queued event in arrival order. Coalesced slots go out
|
|
||||||
// via WriteGSO; passthrough slots go out via plainW.Write. Returns the
|
|
||||||
// first error observed; keeps draining so one bad packet doesn't hold up
|
|
||||||
// the rest. After Flush returns, borrowed payload slices may be recycled.
|
|
||||||
func (c *tcpCoalescer) Flush() error {
|
|
||||||
var first error
|
|
||||||
for _, s := range c.slots {
|
|
||||||
var err error
|
|
||||||
if s.passthrough {
|
|
||||||
_, err = c.plainW.Write(s.rawPkt)
|
|
||||||
} else {
|
|
||||||
err = c.flushSlot(s)
|
|
||||||
}
|
|
||||||
if err != nil && first == nil {
|
|
||||||
first = err
|
|
||||||
}
|
|
||||||
c.release(s)
|
|
||||||
}
|
|
||||||
for i := range c.slots {
|
|
||||||
c.slots[i] = nil
|
|
||||||
}
|
|
||||||
c.slots = c.slots[:0]
|
|
||||||
for k := range c.openSlots {
|
|
||||||
delete(c.openSlots, k)
|
|
||||||
}
|
|
||||||
return first
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *tcpCoalescer) addPassthrough(pkt []byte) {
|
|
||||||
s := c.take()
|
|
||||||
s.passthrough = true
|
|
||||||
s.rawPkt = pkt
|
|
||||||
c.slots = append(c.slots, s)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *tcpCoalescer) seed(pkt []byte, info parsedTCP) {
|
|
||||||
if info.hdrLen > tcpCoalesceHdrCap || info.hdrLen+info.payLen > tcpCoalesceBufSize {
|
|
||||||
// Pathological shape — can't fit our scratch, emit as-is.
|
|
||||||
c.addPassthrough(pkt)
|
|
||||||
return
|
|
||||||
}
|
|
||||||
s := c.take()
|
|
||||||
s.passthrough = false
|
|
||||||
s.rawPkt = nil
|
|
||||||
copy(s.hdrBuf[:], pkt[:info.hdrLen])
|
|
||||||
s.hdrLen = info.hdrLen
|
|
||||||
s.ipHdrLen = info.ipHdrLen
|
|
||||||
s.isV6 = info.fk.isV6
|
|
||||||
s.fk = info.fk
|
|
||||||
s.gsoSize = info.payLen
|
|
||||||
s.numSeg = 1
|
|
||||||
s.totalPay = info.payLen
|
|
||||||
s.nextSeq = info.seq + uint32(info.payLen)
|
|
||||||
s.psh = info.flags&0x08 != 0
|
|
||||||
s.payIovs = append(s.payIovs[:0], pkt[info.hdrLen:info.hdrLen+info.payLen])
|
|
||||||
c.slots = append(c.slots, s)
|
|
||||||
if !s.psh {
|
|
||||||
c.openSlots[info.fk] = s
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// canAppend reports whether info's packet extends the slot's seed: same
|
|
||||||
// header shape and stable contents, adjacent seq, not oversized, chain not
|
|
||||||
// closed.
|
|
||||||
func (c *tcpCoalescer) canAppend(s *coalesceSlot, pkt []byte, info parsedTCP) bool {
|
|
||||||
if s.psh {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if info.hdrLen != s.hdrLen {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if info.seq != s.nextSeq {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if s.numSeg >= tcpCoalesceMaxSegs {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if info.payLen > s.gsoSize {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if s.hdrLen+s.totalPay+info.payLen > tcpCoalesceBufSize {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if !headersMatch(s.hdrBuf[:s.hdrLen], pkt[:info.hdrLen], s.isV6, s.ipHdrLen) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *tcpCoalescer) appendPayload(s *coalesceSlot, pkt []byte, info parsedTCP) {
|
|
||||||
s.payIovs = append(s.payIovs, pkt[info.hdrLen:info.hdrLen+info.payLen])
|
|
||||||
s.numSeg++
|
|
||||||
s.totalPay += info.payLen
|
|
||||||
s.nextSeq = info.seq + uint32(info.payLen)
|
|
||||||
if info.payLen < s.gsoSize || info.flags&0x08 != 0 {
|
|
||||||
s.psh = true
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *tcpCoalescer) take() *coalesceSlot {
|
|
||||||
if n := len(c.pool); n > 0 {
|
|
||||||
s := c.pool[n-1]
|
|
||||||
c.pool[n-1] = nil
|
|
||||||
c.pool = c.pool[:n-1]
|
|
||||||
return s
|
|
||||||
}
|
|
||||||
return &coalesceSlot{}
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *tcpCoalescer) release(s *coalesceSlot) {
|
|
||||||
s.passthrough = false
|
|
||||||
s.rawPkt = nil
|
|
||||||
for i := range s.payIovs {
|
|
||||||
s.payIovs[i] = nil
|
|
||||||
}
|
|
||||||
s.payIovs = s.payIovs[:0]
|
|
||||||
s.numSeg = 0
|
|
||||||
s.totalPay = 0
|
|
||||||
s.psh = false
|
|
||||||
c.pool = append(c.pool, s)
|
|
||||||
}
|
|
||||||
|
|
||||||
// flushSlot patches the header and calls WriteGSO. Does not remove the
|
|
||||||
// slot from c.slots.
|
|
||||||
func (c *tcpCoalescer) flushSlot(s *coalesceSlot) error {
|
|
||||||
total := s.hdrLen + s.totalPay
|
|
||||||
l4Len := total - s.ipHdrLen
|
|
||||||
hdr := s.hdrBuf[:s.hdrLen]
|
|
||||||
|
|
||||||
if s.isV6 {
|
|
||||||
binary.BigEndian.PutUint16(hdr[4:6], uint16(l4Len))
|
|
||||||
} else {
|
|
||||||
binary.BigEndian.PutUint16(hdr[2:4], uint16(total))
|
|
||||||
hdr[10] = 0
|
|
||||||
hdr[11] = 0
|
|
||||||
binary.BigEndian.PutUint16(hdr[10:12], ipv4HdrChecksum(hdr[:s.ipHdrLen]))
|
|
||||||
}
|
|
||||||
|
|
||||||
var psum uint32
|
|
||||||
if s.isV6 {
|
|
||||||
psum = pseudoSumIPv6(hdr[8:24], hdr[24:40], ipProtoTCP, l4Len)
|
|
||||||
} else {
|
|
||||||
psum = pseudoSumIPv4(hdr[12:16], hdr[16:20], ipProtoTCP, l4Len)
|
|
||||||
}
|
|
||||||
tcsum := s.ipHdrLen + 16
|
|
||||||
binary.BigEndian.PutUint16(hdr[tcsum:tcsum+2], foldOnceNoInvert(psum))
|
|
||||||
|
|
||||||
return c.gsoW.WriteGSO(hdr, s.payIovs, uint16(s.gsoSize), s.isV6, uint16(s.ipHdrLen))
|
|
||||||
}
|
|
||||||
|
|
||||||
// headersMatch compares two IP+TCP header prefixes for byte-for-byte
|
|
||||||
// equality on every field that must be identical across coalesced
|
|
||||||
// segments. Size/IPID/IPCsum/seq/flags/tcpCsum are masked out.
|
|
||||||
func headersMatch(a, b []byte, isV6 bool, ipHdrLen int) bool {
|
|
||||||
if len(a) != len(b) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if isV6 {
|
|
||||||
// IPv6: bytes [0:4] = version/TC/flow-label, [6:8] = next_hdr/hop,
|
|
||||||
// [8:40] = src+dst. Skip [4:6] payload length.
|
|
||||||
if !bytes.Equal(a[0:4], b[0:4]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if !bytes.Equal(a[6:40], b[6:40]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
// IPv4: [0:2] version/IHL/TOS, [6:10] flags/fragoff/TTL/proto,
|
|
||||||
// [12:20] src+dst. Skip [2:4] total len, [4:6] id, [10:12] csum.
|
|
||||||
if !bytes.Equal(a[0:2], b[0:2]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if !bytes.Equal(a[6:10], b[6:10]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if !bytes.Equal(a[12:20], b[12:20]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// TCP: compare [0:4] ports, [8:13] ack+dataoff, [14:16] window,
|
|
||||||
// [18:tcpHdrLen] options (incl. urgent).
|
|
||||||
tcp := ipHdrLen
|
|
||||||
if !bytes.Equal(a[tcp:tcp+4], b[tcp:tcp+4]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if !bytes.Equal(a[tcp+8:tcp+13], b[tcp+8:tcp+13]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if !bytes.Equal(a[tcp+14:tcp+16], b[tcp+14:tcp+16]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
if !bytes.Equal(a[tcp+18:], b[tcp+18:]) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
|
|
||||||
// ipv4HdrChecksum computes the IPv4 header checksum over hdr (which must
|
|
||||||
// already have its checksum field zeroed) and returns the folded/inverted
|
|
||||||
// 16-bit value to store.
|
|
||||||
func ipv4HdrChecksum(hdr []byte) uint16 {
|
|
||||||
var sum uint32
|
|
||||||
for i := 0; i+1 < len(hdr); i += 2 {
|
|
||||||
sum += uint32(binary.BigEndian.Uint16(hdr[i : i+2]))
|
|
||||||
}
|
|
||||||
if len(hdr)%2 == 1 {
|
|
||||||
sum += uint32(hdr[len(hdr)-1]) << 8
|
|
||||||
}
|
|
||||||
for sum>>16 != 0 {
|
|
||||||
sum = (sum & 0xffff) + (sum >> 16)
|
|
||||||
}
|
|
||||||
return ^uint16(sum)
|
|
||||||
}
|
|
||||||
|
|
||||||
// pseudoSumIPv4 / pseudoSumIPv6 build the TCP pseudo-header partial sum
|
|
||||||
// expected by the virtio NEEDS_CSUM kernel path: the 32-bit accumulator
|
|
||||||
// before folding.
|
|
||||||
func pseudoSumIPv4(src, dst []byte, proto byte, l4Len int) uint32 {
|
|
||||||
var sum uint32
|
|
||||||
sum += uint32(binary.BigEndian.Uint16(src[0:2]))
|
|
||||||
sum += uint32(binary.BigEndian.Uint16(src[2:4]))
|
|
||||||
sum += uint32(binary.BigEndian.Uint16(dst[0:2]))
|
|
||||||
sum += uint32(binary.BigEndian.Uint16(dst[2:4]))
|
|
||||||
sum += uint32(proto)
|
|
||||||
sum += uint32(l4Len)
|
|
||||||
return sum
|
|
||||||
}
|
|
||||||
|
|
||||||
func pseudoSumIPv6(src, dst []byte, proto byte, l4Len int) uint32 {
|
|
||||||
var sum uint32
|
|
||||||
for i := 0; i < 16; i += 2 {
|
|
||||||
sum += uint32(binary.BigEndian.Uint16(src[i : i+2]))
|
|
||||||
sum += uint32(binary.BigEndian.Uint16(dst[i : i+2]))
|
|
||||||
}
|
|
||||||
sum += uint32(l4Len >> 16)
|
|
||||||
sum += uint32(l4Len & 0xffff)
|
|
||||||
sum += uint32(proto)
|
|
||||||
return sum
|
|
||||||
}
|
|
||||||
|
|
||||||
// foldOnceNoInvert folds the 32-bit accumulator to 16 bits and returns it
|
|
||||||
// unchanged (no one's complement). This is what virtio NEEDS_CSUM wants in
|
|
||||||
// the L4 checksum field — the kernel will add the payload sum and invert.
|
|
||||||
func foldOnceNoInvert(sum uint32) uint16 {
|
|
||||||
for sum>>16 != 0 {
|
|
||||||
sum = (sum & 0xffff) + (sum >> 16)
|
|
||||||
}
|
|
||||||
return uint16(sum)
|
|
||||||
}
|
|
||||||
@@ -1,576 +0,0 @@
|
|||||||
package nebula
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/binary"
|
|
||||||
"testing"
|
|
||||||
)
|
|
||||||
|
|
||||||
// fakeTunWriter records plain Writes and WriteGSO calls without touching a
|
|
||||||
// real TUN fd. WriteGSO preserves the split between hdr and borrowed pays
|
|
||||||
// so tests can inspect each independently.
|
|
||||||
type fakeTunWriter struct {
|
|
||||||
gsoEnabled bool
|
|
||||||
writes [][]byte
|
|
||||||
gsoWrites []fakeGSOWrite
|
|
||||||
}
|
|
||||||
|
|
||||||
type fakeGSOWrite struct {
|
|
||||||
hdr []byte
|
|
||||||
pays [][]byte
|
|
||||||
gsoSize uint16
|
|
||||||
isV6 bool
|
|
||||||
csumStart uint16
|
|
||||||
}
|
|
||||||
|
|
||||||
// total returns hdrLen + sum of pay lens.
|
|
||||||
func (g fakeGSOWrite) total() int {
|
|
||||||
n := len(g.hdr)
|
|
||||||
for _, p := range g.pays {
|
|
||||||
n += len(p)
|
|
||||||
}
|
|
||||||
return n
|
|
||||||
}
|
|
||||||
|
|
||||||
// payLen sums the pays.
|
|
||||||
func (g fakeGSOWrite) payLen() int {
|
|
||||||
var n int
|
|
||||||
for _, p := range g.pays {
|
|
||||||
n += len(p)
|
|
||||||
}
|
|
||||||
return n
|
|
||||||
}
|
|
||||||
|
|
||||||
func (w *fakeTunWriter) Write(p []byte) (int, error) {
|
|
||||||
buf := make([]byte, len(p))
|
|
||||||
copy(buf, p)
|
|
||||||
w.writes = append(w.writes, buf)
|
|
||||||
return len(p), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (w *fakeTunWriter) WriteGSO(hdr []byte, pays [][]byte, gsoSize uint16, isV6 bool, csumStart uint16) error {
|
|
||||||
hcopy := make([]byte, len(hdr))
|
|
||||||
copy(hcopy, hdr)
|
|
||||||
paysCopy := make([][]byte, len(pays))
|
|
||||||
for i, p := range pays {
|
|
||||||
pc := make([]byte, len(p))
|
|
||||||
copy(pc, p)
|
|
||||||
paysCopy[i] = pc
|
|
||||||
}
|
|
||||||
w.gsoWrites = append(w.gsoWrites, fakeGSOWrite{
|
|
||||||
hdr: hcopy,
|
|
||||||
pays: paysCopy,
|
|
||||||
gsoSize: gsoSize,
|
|
||||||
isV6: isV6,
|
|
||||||
csumStart: csumStart,
|
|
||||||
})
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (w *fakeTunWriter) GSOSupported() bool { return w.gsoEnabled }
|
|
||||||
|
|
||||||
// buildTCPv4 constructs a minimal IPv4+TCP packet with the given payload,
|
|
||||||
// seq, and flags. Assumes no IP options and a 20-byte TCP header.
|
|
||||||
func buildTCPv4(seq uint32, flags byte, payload []byte) []byte {
|
|
||||||
return buildTCPv4Ports(1000, 2000, seq, flags, payload)
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildTCPv4Ports is buildTCPv4 with caller-specified ports so tests can
|
|
||||||
// build distinct flows.
|
|
||||||
func buildTCPv4Ports(sport, dport uint16, seq uint32, flags byte, payload []byte) []byte {
|
|
||||||
const ipHdrLen = 20
|
|
||||||
const tcpHdrLen = 20
|
|
||||||
total := ipHdrLen + tcpHdrLen + len(payload)
|
|
||||||
pkt := make([]byte, total)
|
|
||||||
|
|
||||||
pkt[0] = 0x45
|
|
||||||
pkt[1] = 0x00
|
|
||||||
binary.BigEndian.PutUint16(pkt[2:4], uint16(total))
|
|
||||||
binary.BigEndian.PutUint16(pkt[4:6], 0)
|
|
||||||
binary.BigEndian.PutUint16(pkt[6:8], 0x4000)
|
|
||||||
pkt[8] = 64
|
|
||||||
pkt[9] = ipProtoTCP
|
|
||||||
copy(pkt[12:16], []byte{10, 0, 0, 1})
|
|
||||||
copy(pkt[16:20], []byte{10, 0, 0, 2})
|
|
||||||
|
|
||||||
binary.BigEndian.PutUint16(pkt[20:22], sport)
|
|
||||||
binary.BigEndian.PutUint16(pkt[22:24], dport)
|
|
||||||
binary.BigEndian.PutUint32(pkt[24:28], seq)
|
|
||||||
binary.BigEndian.PutUint32(pkt[28:32], 12345)
|
|
||||||
pkt[32] = 0x50
|
|
||||||
pkt[33] = flags
|
|
||||||
binary.BigEndian.PutUint16(pkt[34:36], 0xffff)
|
|
||||||
|
|
||||||
copy(pkt[40:], payload)
|
|
||||||
return pkt
|
|
||||||
}
|
|
||||||
|
|
||||||
const (
|
|
||||||
tcpAck = 0x10
|
|
||||||
tcpPsh = 0x08
|
|
||||||
tcpSyn = 0x02
|
|
||||||
tcpFin = 0x01
|
|
||||||
tcpAckPsh = tcpAck | tcpPsh
|
|
||||||
)
|
|
||||||
|
|
||||||
func TestCoalescerPassthroughWhenGSOUnavailable(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: false}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pkt := buildTCPv4(1000, tcpAck, []byte("hello"))
|
|
||||||
if err := c.Add(pkt); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// No sync write — passthrough is deferred to Flush.
|
|
||||||
if len(w.writes) != 0 || len(w.gsoWrites) != 0 {
|
|
||||||
t.Fatalf("no Add-time writes: got writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if len(w.writes) != 1 || len(w.gsoWrites) != 0 {
|
|
||||||
t.Fatalf("want single plain write, got writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerNonTCPPassthrough(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pkt := make([]byte, 28)
|
|
||||||
pkt[0] = 0x45
|
|
||||||
binary.BigEndian.PutUint16(pkt[2:4], 28)
|
|
||||||
pkt[9] = 1
|
|
||||||
copy(pkt[12:16], []byte{10, 0, 0, 1})
|
|
||||||
copy(pkt[16:20], []byte{10, 0, 0, 2})
|
|
||||||
if err := c.Add(pkt); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if len(w.writes) != 1 || len(w.gsoWrites) != 0 {
|
|
||||||
t.Fatalf("ICMP should pass through unchanged")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerSeedThenFlushAlone(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pkt := buildTCPv4(1000, tcpAck, make([]byte, 1000))
|
|
||||||
if err := c.Add(pkt); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if len(w.writes) != 0 || len(w.gsoWrites) != 0 {
|
|
||||||
t.Fatalf("unexpected output before flush")
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Single-segment flush now goes through WriteGSO with GSO_NONE
|
|
||||||
// (virtio NEEDS_CSUM lets the kernel fill in the L4 csum).
|
|
||||||
if len(w.gsoWrites) != 1 || len(w.writes) != 0 {
|
|
||||||
t.Fatalf("single-seg flush: writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
g := w.gsoWrites[0]
|
|
||||||
if g.total() != 40+1000 {
|
|
||||||
t.Errorf("super total=%d want %d", g.total(), 40+1000)
|
|
||||||
}
|
|
||||||
if g.payLen() != 1000 {
|
|
||||||
t.Errorf("payLen=%d want 1000", g.payLen())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerCoalescesAdjacentACKs(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
if err := c.Add(buildTCPv4(1000, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4(2200, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4(3400, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if len(w.gsoWrites) != 1 {
|
|
||||||
t.Fatalf("want 1 gso write, got %d (plain=%d)", len(w.gsoWrites), len(w.writes))
|
|
||||||
}
|
|
||||||
g := w.gsoWrites[0]
|
|
||||||
if g.gsoSize != 1200 {
|
|
||||||
t.Errorf("gsoSize=%d want 1200", g.gsoSize)
|
|
||||||
}
|
|
||||||
if len(g.hdr) != 40 {
|
|
||||||
t.Errorf("hdrLen=%d want 40", len(g.hdr))
|
|
||||||
}
|
|
||||||
if g.csumStart != 20 {
|
|
||||||
t.Errorf("csumStart=%d want 20", g.csumStart)
|
|
||||||
}
|
|
||||||
if len(g.pays) != 3 {
|
|
||||||
t.Errorf("pay count=%d want 3", len(g.pays))
|
|
||||||
}
|
|
||||||
if g.total() != 40+3*1200 {
|
|
||||||
t.Errorf("superpacket len=%d want %d", g.total(), 40+3*1200)
|
|
||||||
}
|
|
||||||
if tot := binary.BigEndian.Uint16(g.hdr[2:4]); int(tot) != g.total() {
|
|
||||||
t.Errorf("ip total_length=%d want %d", tot, g.total())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerRejectsSeqGap(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
if err := c.Add(buildTCPv4(1000, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4(3000, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Each packet flushes as its own single-segment WriteGSO now.
|
|
||||||
if len(w.gsoWrites) != 2 || len(w.writes) != 0 {
|
|
||||||
t.Fatalf("seq gap: want 2 gso writes got writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerRejectsFlagMismatch(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
if err := c.Add(buildTCPv4(1000, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// SYN|ACK is non-admissible. Must flush matching flow's slot (gso)
|
|
||||||
// and then plain-write the SYN packet itself.
|
|
||||||
syn := buildTCPv4(2200, tcpSyn|tcpAck, pay)
|
|
||||||
if err := c.Add(syn); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if len(w.writes) != 1 || len(w.gsoWrites) != 1 {
|
|
||||||
t.Fatalf("flag mismatch: want 1 plain + 1 gso, got writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerRejectsFIN(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
fin := buildTCPv4(1000, tcpAck|tcpFin, []byte("x"))
|
|
||||||
if err := c.Add(fin); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// FIN isn't admissible — passthrough as plain, no slot, no gso.
|
|
||||||
if len(w.writes) != 1 || len(w.gsoWrites) != 0 {
|
|
||||||
t.Fatalf("FIN should be passthrough, got writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerShortLastSegmentClosesChain(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
full := make([]byte, 1200)
|
|
||||||
half := make([]byte, 500)
|
|
||||||
if err := c.Add(buildTCPv4(1000, tcpAck, full)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4(2200, tcpAck, half)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Chain now closed; next packet seeds a new slot on the same flow
|
|
||||||
// after flushing the old one.
|
|
||||||
if err := c.Add(buildTCPv4(2700, tcpAck, full)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Expect two gso writes: the first two packets coalesced, then the
|
|
||||||
// third flushed alone (single-seg via GSO_NONE).
|
|
||||||
if len(w.gsoWrites) != 2 {
|
|
||||||
t.Fatalf("want 2 gso writes got %d", len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
if len(w.writes) != 0 {
|
|
||||||
t.Fatalf("want 0 plain writes got %d", len(w.writes))
|
|
||||||
}
|
|
||||||
if w.gsoWrites[0].gsoSize != 1200 {
|
|
||||||
t.Errorf("gsoSize=%d want 1200", w.gsoWrites[0].gsoSize)
|
|
||||||
}
|
|
||||||
if got, want := w.gsoWrites[0].total(), 40+1200+500; got != want {
|
|
||||||
t.Errorf("super len=%d want %d", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerPSHFinalizesChain(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
if err := c.Add(buildTCPv4(1000, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4(2200, tcpAckPsh, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4(3400, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// First two coalesce; the third seeds a fresh slot that flushes alone.
|
|
||||||
if len(w.gsoWrites) != 2 {
|
|
||||||
t.Fatalf("want 2 gso writes got %d", len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
if len(w.writes) != 0 {
|
|
||||||
t.Fatalf("want 0 plain writes got %d", len(w.writes))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerRejectsDifferentFlow(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
p1 := buildTCPv4(1000, tcpAck, pay)
|
|
||||||
p2 := buildTCPv4(2200, tcpAck, pay)
|
|
||||||
binary.BigEndian.PutUint16(p2[20:22], 9999)
|
|
||||||
if err := c.Add(p1); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(p2); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Two independent flows, each flushes its own single-segment WriteGSO.
|
|
||||||
if len(w.gsoWrites) != 2 || len(w.writes) != 0 {
|
|
||||||
t.Fatalf("diff flow: want 2 gso writes got writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerRejectsIPOptions(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 500)
|
|
||||||
pkt := buildTCPv4(1000, tcpAck, pay)
|
|
||||||
// Bump IHL to 6 to simulate 4 bytes of IP options. Don't actually add
|
|
||||||
// bytes — parser should bail before it matters.
|
|
||||||
pkt[0] = 0x46
|
|
||||||
if err := c.Add(pkt); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Non-admissible parse → passthrough as plain.
|
|
||||||
if len(w.writes) != 1 || len(w.gsoWrites) != 0 {
|
|
||||||
t.Fatalf("IP options should passthrough, got writes=%d gso=%d", len(w.writes), len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCoalescerCapBySegments(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 512)
|
|
||||||
seq := uint32(1000)
|
|
||||||
for i := 0; i < tcpCoalesceMaxSegs+5; i++ {
|
|
||||||
if err := c.Add(buildTCPv4(seq, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
seq += uint32(len(pay))
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
for _, g := range w.gsoWrites {
|
|
||||||
segs := len(g.pays)
|
|
||||||
if segs > tcpCoalesceMaxSegs {
|
|
||||||
t.Fatalf("super exceeded seg cap: %d > %d", segs, tcpCoalesceMaxSegs)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestCoalescerMultipleFlowsInSameBatch proves two interleaved bulk TCP
|
|
||||||
// flows coalesce independently in a single Flush.
|
|
||||||
func TestCoalescerMultipleFlowsInSameBatch(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
|
|
||||||
// Flow A: sport 1000. Flow B: sport 3000.
|
|
||||||
if err := c.Add(buildTCPv4Ports(1000, 2000, 100, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(3000, 2000, 500, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(1000, 2000, 1300, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(3000, 2000, 1700, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(1000, 2000, 2500, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(3000, 2000, 2900, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(w.gsoWrites) != 2 {
|
|
||||||
t.Fatalf("want 2 gso writes (one per flow), got %d", len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
if len(w.writes) != 0 {
|
|
||||||
t.Fatalf("want no plain writes, got %d", len(w.writes))
|
|
||||||
}
|
|
||||||
// Each superpacket should carry 3 segments.
|
|
||||||
for i, g := range w.gsoWrites {
|
|
||||||
if len(g.pays) != 3 {
|
|
||||||
t.Errorf("gso[%d]: segs=%d want 3", i, len(g.pays))
|
|
||||||
}
|
|
||||||
if g.gsoSize != 1200 {
|
|
||||||
t.Errorf("gso[%d]: gsoSize=%d want 1200", i, g.gsoSize)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Verify each superpacket carries the source port it was seeded with.
|
|
||||||
seenSports := map[uint16]bool{}
|
|
||||||
for _, g := range w.gsoWrites {
|
|
||||||
sp := binary.BigEndian.Uint16(g.hdr[20:22])
|
|
||||||
seenSports[sp] = true
|
|
||||||
}
|
|
||||||
if !seenSports[1000] || !seenSports[3000] {
|
|
||||||
t.Errorf("expected superpackets for sports 1000 and 3000, got %v", seenSports)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestCoalescerPreservesArrivalOrder confirms that with passthrough and
|
|
||||||
// coalesced events both queued, Flush emits them in Add order rather than
|
|
||||||
// writing passthrough packets synchronously.
|
|
||||||
func TestCoalescerPreservesArrivalOrder(t *testing.T) {
|
|
||||||
w := &orderedFakeWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
// Sequence: coalesceable TCP, ICMP (passthrough), coalesceable TCP on
|
|
||||||
// a different flow. Expected emit order: gso(X), plain(ICMP), gso(Y).
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
if err := c.Add(buildTCPv4Ports(1000, 2000, 100, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
icmp := make([]byte, 28)
|
|
||||||
icmp[0] = 0x45
|
|
||||||
binary.BigEndian.PutUint16(icmp[2:4], 28)
|
|
||||||
icmp[9] = 1
|
|
||||||
copy(icmp[12:16], []byte{10, 0, 0, 1})
|
|
||||||
copy(icmp[16:20], []byte{10, 0, 0, 3})
|
|
||||||
if err := c.Add(icmp); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(3000, 2000, 500, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Nothing should have hit the writer synchronously.
|
|
||||||
if len(w.events) != 0 {
|
|
||||||
t.Fatalf("Add emitted events synchronously: %v", w.events)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if got, want := w.events, []string{"gso", "plain", "gso"}; !stringSliceEq(got, want) {
|
|
||||||
t.Fatalf("flush order=%v want %v", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// orderedFakeWriter records only the sequence of call types so tests can
|
|
||||||
// assert arrival order without inspecting bytes.
|
|
||||||
type orderedFakeWriter struct {
|
|
||||||
gsoEnabled bool
|
|
||||||
events []string
|
|
||||||
}
|
|
||||||
|
|
||||||
func (w *orderedFakeWriter) Write(p []byte) (int, error) {
|
|
||||||
w.events = append(w.events, "plain")
|
|
||||||
return len(p), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (w *orderedFakeWriter) WriteGSO(hdr []byte, pays [][]byte, gsoSize uint16, isV6 bool, csumStart uint16) error {
|
|
||||||
w.events = append(w.events, "gso")
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (w *orderedFakeWriter) GSOSupported() bool { return w.gsoEnabled }
|
|
||||||
|
|
||||||
func stringSliceEq(a, b []string) bool {
|
|
||||||
if len(a) != len(b) {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
for i := range a {
|
|
||||||
if a[i] != b[i] {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestCoalescerInterleavedFlowsPreserveOrdering checks that a non-admissible
|
|
||||||
// packet (SYN) mid-flow only flushes its own flow, not others.
|
|
||||||
func TestCoalescerInterleavedFlowsPreserveOrdering(t *testing.T) {
|
|
||||||
w := &fakeTunWriter{gsoEnabled: true}
|
|
||||||
c := newTCPCoalescer(w)
|
|
||||||
pay := make([]byte, 1200)
|
|
||||||
|
|
||||||
// Flow A two segments.
|
|
||||||
if err := c.Add(buildTCPv4Ports(1000, 2000, 100, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(1000, 2000, 1300, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Flow B two segments.
|
|
||||||
if err := c.Add(buildTCPv4Ports(3000, 2000, 500, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Add(buildTCPv4Ports(3000, 2000, 1700, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Flow A SYN (non-admissible) — must flush only flow A's slot.
|
|
||||||
syn := buildTCPv4Ports(1000, 2000, 9999, tcpSyn|tcpAck, pay)
|
|
||||||
if err := c.Add(syn); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
// Flow B continues — should still be coalesced with its seed.
|
|
||||||
if err := c.Add(buildTCPv4Ports(3000, 2000, 2900, tcpAck, pay)); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if err := c.Flush(); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Expected:
|
|
||||||
// - 1 gso for flow A (first 2 segments)
|
|
||||||
// - 1 plain for flow A SYN
|
|
||||||
// - 1 gso for flow B (3 segments)
|
|
||||||
if len(w.gsoWrites) != 2 {
|
|
||||||
t.Fatalf("want 2 gso writes, got %d", len(w.gsoWrites))
|
|
||||||
}
|
|
||||||
if len(w.writes) != 1 {
|
|
||||||
t.Fatalf("want 1 plain write (SYN), got %d", len(w.writes))
|
|
||||||
}
|
|
||||||
// Find the 3-segment gso (flow B) and the 2-segment gso (flow A).
|
|
||||||
var segCounts []int
|
|
||||||
for _, g := range w.gsoWrites {
|
|
||||||
segCounts = append(segCounts, len(g.pays))
|
|
||||||
}
|
|
||||||
if !(segCounts[0] == 2 && segCounts[1] == 3) && !(segCounts[0] == 3 && segCounts[1] == 2) {
|
|
||||||
t.Errorf("unexpected segment counts: %v (want 2 and 3)", segCounts)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,7 +1,8 @@
|
|||||||
package overlay
|
package test
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"errors"
|
"errors"
|
||||||
|
"io"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
|
|
||||||
"github.com/slackhq/nebula/routing"
|
"github.com/slackhq/nebula/routing"
|
||||||
@@ -25,15 +26,11 @@ func (NoopTun) Name() string {
|
|||||||
return "noop"
|
return "noop"
|
||||||
}
|
}
|
||||||
|
|
||||||
func (NoopTun) Read() ([][]byte, error) {
|
func (NoopTun) Read([]byte) (int, error) {
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (NoopTun) Write([]byte) (int, error) {
|
|
||||||
return 0, nil
|
return 0, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (NoopTun) WriteReject(p []byte) (int, error) {
|
func (NoopTun) Write([]byte) (int, error) {
|
||||||
return 0, nil
|
return 0, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -41,7 +38,7 @@ func (NoopTun) SupportsMultiqueue() bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (NoopTun) NewMultiQueueReader() (Queue, error) {
|
func (NoopTun) NewMultiQueueReader() (io.ReadWriteCloser, error) {
|
||||||
return nil, errors.New("unsupported")
|
return nil, errors.New("unsupported")
|
||||||
}
|
}
|
||||||
|
|
||||||
+2
-38
@@ -8,12 +8,6 @@ import (
|
|||||||
|
|
||||||
const MTU = 9001
|
const MTU = 9001
|
||||||
|
|
||||||
// MaxWriteBatch is the largest batch any Conn.WriteBatch implementation is
|
|
||||||
// required to accept. Callers SHOULD NOT pass more than this per call; Linux
|
|
||||||
// backends preallocate sendmmsg scratch sized to this value, so exceeding it
|
|
||||||
// only costs a chunked retry.
|
|
||||||
const MaxWriteBatch = 128
|
|
||||||
|
|
||||||
type EncReader func(
|
type EncReader func(
|
||||||
addr netip.AddrPort,
|
addr netip.AddrPort,
|
||||||
payload []byte,
|
payload []byte,
|
||||||
@@ -22,29 +16,8 @@ type EncReader func(
|
|||||||
type Conn interface {
|
type Conn interface {
|
||||||
Rebind() error
|
Rebind() error
|
||||||
LocalAddr() (netip.AddrPort, error)
|
LocalAddr() (netip.AddrPort, error)
|
||||||
// ListenOut invokes r for each received packet. On batch-capable
|
ListenOut(r EncReader) error
|
||||||
// backends (recvmmsg), flush is called after each batch is fully
|
|
||||||
// delivered — callers use it to flush per-batch accumulators such as
|
|
||||||
// TUN write coalescers. Single-packet backends call flush after each
|
|
||||||
// packet. flush must not be nil.
|
|
||||||
ListenOut(r EncReader, flush func()) error
|
|
||||||
WriteTo(b []byte, addr netip.AddrPort) error
|
WriteTo(b []byte, addr netip.AddrPort) error
|
||||||
// WriteBatch sends a contiguous batch of packets, each with its own
|
|
||||||
// destination. bufs and addrs must have the same length. Linux uses
|
|
||||||
// sendmmsg(2) for a single syscall; other backends fall back to a
|
|
||||||
// WriteTo loop. Returns on the first error; callers may observe a
|
|
||||||
// partial send if some packets went out before the error.
|
|
||||||
WriteBatch(bufs [][]byte, addrs []netip.AddrPort) error
|
|
||||||
// WriteSegmented sends bufs as a single UDP GSO sendmsg when the kernel
|
|
||||||
// supports it: all bufs go to the same addr, each must be exactly segSize
|
|
||||||
// bytes except the last which may be shorter. The kernel emits one
|
|
||||||
// datagram per buf on the wire. Backends / kernels without GSO support
|
|
||||||
// fall back to a per-packet WriteTo loop. Returns on the first error.
|
|
||||||
WriteSegmented(bufs [][]byte, addr netip.AddrPort, segSize int) error
|
|
||||||
// SupportsGSO reports whether WriteSegmented takes the single-syscall
|
|
||||||
// GSO path. Callers use this to decide at batch-assembly time whether
|
|
||||||
// the uniform-size / same-dst check is worth running.
|
|
||||||
SupportsGSO() bool
|
|
||||||
ReloadConfig(c *config.C)
|
ReloadConfig(c *config.C)
|
||||||
SupportsMultipleReaders() bool
|
SupportsMultipleReaders() bool
|
||||||
Close() error
|
Close() error
|
||||||
@@ -58,7 +31,7 @@ func (NoopConn) Rebind() error {
|
|||||||
func (NoopConn) LocalAddr() (netip.AddrPort, error) {
|
func (NoopConn) LocalAddr() (netip.AddrPort, error) {
|
||||||
return netip.AddrPort{}, nil
|
return netip.AddrPort{}, nil
|
||||||
}
|
}
|
||||||
func (NoopConn) ListenOut(_ EncReader, _ func()) error {
|
func (NoopConn) ListenOut(_ EncReader) error {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
func (NoopConn) SupportsMultipleReaders() bool {
|
func (NoopConn) SupportsMultipleReaders() bool {
|
||||||
@@ -67,15 +40,6 @@ func (NoopConn) SupportsMultipleReaders() bool {
|
|||||||
func (NoopConn) WriteTo(_ []byte, _ netip.AddrPort) error {
|
func (NoopConn) WriteTo(_ []byte, _ netip.AddrPort) error {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
func (NoopConn) WriteBatch(_ [][]byte, _ []netip.AddrPort) error {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
func (NoopConn) WriteSegmented(_ [][]byte, _ netip.AddrPort, _ int) error {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
func (NoopConn) SupportsGSO() bool {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
func (NoopConn) ReloadConfig(_ *config.C) {
|
func (NoopConn) ReloadConfig(_ *config.C) {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-22
@@ -140,26 +140,6 @@ func (u *StdConn) WriteTo(b []byte, ap netip.AddrPort) error {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *StdConn) WriteBatch(bufs [][]byte, addrs []netip.AddrPort) error {
|
|
||||||
for i, b := range bufs {
|
|
||||||
if err := u.WriteTo(b, addrs[i]); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *StdConn) WriteSegmented(bufs [][]byte, addr netip.AddrPort, _ int) error {
|
|
||||||
for _, b := range bufs {
|
|
||||||
if err := u.WriteTo(b, addr); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *StdConn) SupportsGSO() bool { return false }
|
|
||||||
|
|
||||||
func (u *StdConn) LocalAddr() (netip.AddrPort, error) {
|
func (u *StdConn) LocalAddr() (netip.AddrPort, error) {
|
||||||
a := u.UDPConn.LocalAddr()
|
a := u.UDPConn.LocalAddr()
|
||||||
|
|
||||||
@@ -185,7 +165,7 @@ func NewUDPStatsEmitter(udpConns []Conn) func() {
|
|||||||
return func() {}
|
return func() {}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *StdConn) ListenOut(r EncReader, flush func()) error {
|
func (u *StdConn) ListenOut(r EncReader) error {
|
||||||
buffer := make([]byte, MTU)
|
buffer := make([]byte, MTU)
|
||||||
|
|
||||||
for {
|
for {
|
||||||
@@ -200,7 +180,6 @@ func (u *StdConn) ListenOut(r EncReader, flush func()) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
r(netip.AddrPortFrom(rua.Addr().Unmap(), rua.Port()), buffer[:n])
|
r(netip.AddrPortFrom(rua.Addr().Unmap(), rua.Port()), buffer[:n])
|
||||||
flush()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+1
-22
@@ -44,26 +44,6 @@ func (u *GenericConn) WriteTo(b []byte, addr netip.AddrPort) error {
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *GenericConn) WriteBatch(bufs [][]byte, addrs []netip.AddrPort) error {
|
|
||||||
for i, b := range bufs {
|
|
||||||
if _, err := u.UDPConn.WriteToUDPAddrPort(b, addrs[i]); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *GenericConn) WriteSegmented(bufs [][]byte, addr netip.AddrPort, _ int) error {
|
|
||||||
for _, b := range bufs {
|
|
||||||
if _, err := u.UDPConn.WriteToUDPAddrPort(b, addr); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *GenericConn) SupportsGSO() bool { return false }
|
|
||||||
|
|
||||||
func (u *GenericConn) LocalAddr() (netip.AddrPort, error) {
|
func (u *GenericConn) LocalAddr() (netip.AddrPort, error) {
|
||||||
a := u.UDPConn.LocalAddr()
|
a := u.UDPConn.LocalAddr()
|
||||||
|
|
||||||
@@ -93,7 +73,7 @@ type rawMessage struct {
|
|||||||
Len uint32
|
Len uint32
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *GenericConn) ListenOut(r EncReader, flush func()) error {
|
func (u *GenericConn) ListenOut(r EncReader) error {
|
||||||
buffer := make([]byte, MTU)
|
buffer := make([]byte, MTU)
|
||||||
|
|
||||||
var lastRecvErr time.Time
|
var lastRecvErr time.Time
|
||||||
@@ -114,7 +94,6 @@ func (u *GenericConn) ListenOut(r EncReader, flush func()) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
r(netip.AddrPortFrom(rua.Addr().Unmap(), rua.Port()), buffer[:n])
|
r(netip.AddrPortFrom(rua.Addr().Unmap(), rua.Port()), buffer[:n])
|
||||||
flush()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+5
-318
@@ -24,32 +24,6 @@ type StdConn struct {
|
|||||||
isV4 bool
|
isV4 bool
|
||||||
l *logrus.Logger
|
l *logrus.Logger
|
||||||
batch int
|
batch int
|
||||||
|
|
||||||
// sendmmsg scratch. Each queue has its own StdConn, so no locking is
|
|
||||||
// needed. Sized to MaxWriteBatch at construction; WriteBatch chunks
|
|
||||||
// larger inputs.
|
|
||||||
writeMsgs []rawMessage
|
|
||||||
writeIovs []iovec
|
|
||||||
writeNames [][]byte
|
|
||||||
|
|
||||||
// Preallocated closure + in/out slots for sendmmsg, so the hot path
|
|
||||||
// does not heap-allocate a fresh closure per call.
|
|
||||||
writeChunk int
|
|
||||||
writeSent int
|
|
||||||
writeErrno syscall.Errno
|
|
||||||
writeFunc func(fd uintptr) bool
|
|
||||||
|
|
||||||
// UDP GSO (sendmsg with UDP_SEGMENT cmsg) support. gsoSupported is
|
|
||||||
// probed once at socket creation. When true, WriteSegmented takes a
|
|
||||||
// single-syscall GSO path; otherwise it falls back to a WriteTo loop.
|
|
||||||
gsoSupported bool
|
|
||||||
gsoMsg msghdr
|
|
||||||
gsoIovs []iovec
|
|
||||||
gsoName []byte // SizeofSockaddrInet6
|
|
||||||
gsoCmsg []byte // CmsgSpace(2)
|
|
||||||
gsoSent int
|
|
||||||
gsoErrno syscall.Errno
|
|
||||||
gsoFunc func(fd uintptr) bool
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func setReusePort(network, address string, c syscall.RawConn) error {
|
func setReusePort(network, address string, c syscall.RawConn) error {
|
||||||
@@ -96,61 +70,9 @@ func NewListener(l *logrus.Logger, ip netip.Addr, port int, multi bool, batch in
|
|||||||
}
|
}
|
||||||
out.isV4 = af == unix.AF_INET
|
out.isV4 = af == unix.AF_INET
|
||||||
|
|
||||||
out.prepareWriteMessages(MaxWriteBatch)
|
|
||||||
out.writeFunc = out.sendmmsgRawWrite
|
|
||||||
|
|
||||||
out.prepareGSO()
|
|
||||||
|
|
||||||
return out, nil
|
return out, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// maxGSOSegments caps the per-sendmsg GSO fan-out. Linux kernels have
|
|
||||||
// historically capped UDP_MAX_SEGMENTS at 64; newer kernels raise it to 128
|
|
||||||
// but we stay conservative so the same code works everywhere.
|
|
||||||
const maxGSOSegments = 64
|
|
||||||
|
|
||||||
// maxGSOBytes bounds the total payload per sendmsg() when UDP_SEGMENT is
|
|
||||||
// set. The kernel stitches all iovecs into a single skb whose length the
|
|
||||||
// UDP length field can represent, and also enforces sk_gso_max_size (which
|
|
||||||
// on most devices is 65536). We use 65535 so ciphertext + headers always
|
|
||||||
// fits, avoiding EMSGSIZE on large TSO superpackets.
|
|
||||||
const maxGSOBytes = 65535
|
|
||||||
|
|
||||||
// prepareGSO probes UDP_SEGMENT support and, on success, sets up the
|
|
||||||
// reusable sendmsg scratch (iovecs, sockaddr, cmsg) plus the preallocated
|
|
||||||
// raw-write closure used to avoid heap allocations on the hot path.
|
|
||||||
func (u *StdConn) prepareGSO() {
|
|
||||||
var probeErr error
|
|
||||||
if err := u.rawConn.Control(func(fd uintptr) {
|
|
||||||
probeErr = unix.SetsockoptInt(int(fd), unix.IPPROTO_UDP, unix.UDP_SEGMENT, 0)
|
|
||||||
}); err != nil {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if probeErr != nil {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
u.gsoSupported = true
|
|
||||||
u.gsoIovs = make([]iovec, maxGSOSegments)
|
|
||||||
u.gsoName = make([]byte, unix.SizeofSockaddrInet6)
|
|
||||||
u.gsoCmsg = make([]byte, unix.CmsgSpace(2))
|
|
||||||
|
|
||||||
// Wire up the static pieces of gsoMsg. Iovlen / Controllen / Namelen /
|
|
||||||
// cmsg contents get refreshed per call; Iov, Name, Control pointers are
|
|
||||||
// fixed because the scratch slices never move.
|
|
||||||
u.gsoMsg.Iov = &u.gsoIovs[0]
|
|
||||||
u.gsoMsg.Name = &u.gsoName[0]
|
|
||||||
u.gsoMsg.Control = &u.gsoCmsg[0]
|
|
||||||
|
|
||||||
// Prepopulate the cmsg header. Len/Level/Type are constant for our use;
|
|
||||||
// only the 2-byte gso_size payload changes per call.
|
|
||||||
cmsghdr := (*unix.Cmsghdr)(unsafe.Pointer(&u.gsoCmsg[0]))
|
|
||||||
cmsghdr.Level = unix.SOL_UDP
|
|
||||||
cmsghdr.Type = unix.UDP_SEGMENT
|
|
||||||
setCmsgLen(cmsghdr, unix.CmsgLen(2))
|
|
||||||
|
|
||||||
u.gsoFunc = u.sendmsgRawWriteGSO
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *StdConn) SupportsMultipleReaders() bool {
|
func (u *StdConn) SupportsMultipleReaders() bool {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
@@ -249,7 +171,7 @@ func recvmmsg(fd uintptr, msgs []rawMessage) (int, bool, error) {
|
|||||||
return int(n), true, nil
|
return int(n), true, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *StdConn) listenOutSingle(r EncReader, flush func()) error {
|
func (u *StdConn) listenOutSingle(r EncReader) error {
|
||||||
var err error
|
var err error
|
||||||
var n int
|
var n int
|
||||||
var from netip.AddrPort
|
var from netip.AddrPort
|
||||||
@@ -262,11 +184,10 @@ func (u *StdConn) listenOutSingle(r EncReader, flush func()) error {
|
|||||||
}
|
}
|
||||||
from = netip.AddrPortFrom(from.Addr().Unmap(), from.Port())
|
from = netip.AddrPortFrom(from.Addr().Unmap(), from.Port())
|
||||||
r(from, buffer[:n])
|
r(from, buffer[:n])
|
||||||
flush()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *StdConn) listenOutBatch(r EncReader, flush func()) error {
|
func (u *StdConn) listenOutBatch(r EncReader) error {
|
||||||
var ip netip.Addr
|
var ip netip.Addr
|
||||||
var n int
|
var n int
|
||||||
var operr error
|
var operr error
|
||||||
@@ -298,17 +219,14 @@ func (u *StdConn) listenOutBatch(r EncReader, flush func()) error {
|
|||||||
}
|
}
|
||||||
r(netip.AddrPortFrom(ip.Unmap(), binary.BigEndian.Uint16(names[i][2:4])), buffers[i][:msgs[i].Len])
|
r(netip.AddrPortFrom(ip.Unmap(), binary.BigEndian.Uint16(names[i][2:4])), buffers[i][:msgs[i].Len])
|
||||||
}
|
}
|
||||||
// End-of-batch: let callers (e.g. TUN write coalescer) flush any
|
|
||||||
// state they accumulated across this batch.
|
|
||||||
flush()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *StdConn) ListenOut(r EncReader, flush func()) error {
|
func (u *StdConn) ListenOut(r EncReader) error {
|
||||||
if u.batch == 1 {
|
if u.batch == 1 {
|
||||||
return u.listenOutSingle(r, flush)
|
return u.listenOutSingle(r)
|
||||||
} else {
|
} else {
|
||||||
return u.listenOutBatch(r, flush)
|
return u.listenOutBatch(r)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -317,237 +235,6 @@ func (u *StdConn) WriteTo(b []byte, ip netip.AddrPort) error {
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
// WriteBatch sends bufs via sendmmsg(2) using the preallocated scratch on
|
|
||||||
// StdConn. Chunks larger than the scratch are processed in multiple syscalls.
|
|
||||||
// If sendmmsg returns a fatal error mid-chunk we fall back to single WriteTo
|
|
||||||
// calls for the remainder so the caller still gets best-effort delivery.
|
|
||||||
func (u *StdConn) WriteBatch(bufs [][]byte, addrs []netip.AddrPort) error {
|
|
||||||
if len(bufs) != len(addrs) {
|
|
||||||
return fmt.Errorf("WriteBatch: len(bufs)=%d != len(addrs)=%d", len(bufs), len(addrs))
|
|
||||||
}
|
|
||||||
//u.l.WithField("bufs", len(bufs)).Info("WriteBatch")
|
|
||||||
i := 0
|
|
||||||
for i < len(bufs) {
|
|
||||||
chunk := len(bufs) - i
|
|
||||||
if chunk > len(u.writeMsgs) {
|
|
||||||
chunk = len(u.writeMsgs)
|
|
||||||
}
|
|
||||||
|
|
||||||
for k := 0; k < chunk; k++ {
|
|
||||||
b := bufs[i+k]
|
|
||||||
if len(b) == 0 {
|
|
||||||
// sendmmsg with an empty iovec is legal but pointless; fall
|
|
||||||
// through after filling the slot so Base is still valid.
|
|
||||||
u.writeIovs[k].Base = nil
|
|
||||||
setIovLen(&u.writeIovs[k], 0)
|
|
||||||
} else {
|
|
||||||
u.writeIovs[k].Base = &b[0]
|
|
||||||
setIovLen(&u.writeIovs[k], len(b))
|
|
||||||
}
|
|
||||||
nlen, err := writeSockaddr(u.writeNames[k], addrs[i+k], u.isV4)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
u.writeMsgs[k].Hdr.Namelen = uint32(nlen)
|
|
||||||
}
|
|
||||||
|
|
||||||
sent, serr := u.sendmmsg(chunk)
|
|
||||||
if serr != nil {
|
|
||||||
if sent <= 0 {
|
|
||||||
// nothing went out; fall back to WriteTo for this chunk.
|
|
||||||
for k := 0; k < chunk; k++ {
|
|
||||||
if err := u.WriteTo(bufs[i+k], addrs[i+k]); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
i += chunk
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
// partial: treat as success for the sent packets and retry the
|
|
||||||
// remainder on the next outer-loop iteration.
|
|
||||||
}
|
|
||||||
if sent == 0 {
|
|
||||||
return fmt.Errorf("sendmmsg made no progress")
|
|
||||||
}
|
|
||||||
i += sent
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// sendmmsgRawWrite is the preallocated callback passed to rawConn.Write. It
|
|
||||||
// reads its input (u.writeChunk) and writes its outputs (u.writeSent,
|
|
||||||
// u.writeErrno) through StdConn fields so the closure itself does not
|
|
||||||
// capture per-call locals and therefore does not heap-allocate.
|
|
||||||
func (u *StdConn) sendmmsgRawWrite(fd uintptr) bool {
|
|
||||||
r1, _, errno := unix.Syscall6(
|
|
||||||
unix.SYS_SENDMMSG,
|
|
||||||
fd,
|
|
||||||
uintptr(unsafe.Pointer(&u.writeMsgs[0])),
|
|
||||||
uintptr(u.writeChunk),
|
|
||||||
0,
|
|
||||||
0,
|
|
||||||
0,
|
|
||||||
)
|
|
||||||
if errno == syscall.EAGAIN || errno == syscall.EWOULDBLOCK {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
u.writeSent = int(r1)
|
|
||||||
u.writeErrno = errno
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *StdConn) SupportsGSO() bool {
|
|
||||||
return u.gsoSupported
|
|
||||||
}
|
|
||||||
|
|
||||||
// WriteSegmented sends bufs to addr as a UDP GSO superpacket. The kernel
|
|
||||||
// emits one datagram per iovec on the wire; all iovecs except the last must
|
|
||||||
// be exactly segSize bytes. Non-GSO kernels hit the WriteTo fallback.
|
|
||||||
// Called with len(bufs) >= 1. len(bufs) > maxGSOSegments is chunked.
|
|
||||||
func (u *StdConn) WriteSegmented(bufs [][]byte, addr netip.AddrPort, segSize int) error {
|
|
||||||
if len(bufs) == 0 {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
if !u.gsoSupported {
|
|
||||||
for _, b := range bufs {
|
|
||||||
if err := u.WriteTo(b, addr); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
nlen, err := writeSockaddr(u.gsoName, addr, u.isV4)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
u.gsoMsg.Namelen = uint32(nlen)
|
|
||||||
setMsgControllen(&u.gsoMsg, unix.CmsgSpace(2))
|
|
||||||
|
|
||||||
// Cap the per-syscall fan-out by both segment count and total bytes.
|
|
||||||
// Kernel rejects sendmsg with EMSGSIZE when segCount*segSize would
|
|
||||||
// exceed sk_gso_max_size (typically 65536). For segSize > maxGSOBytes
|
|
||||||
// we can't use GSO at all and must fall back per-packet.
|
|
||||||
segsByBytes := maxGSOBytes / segSize
|
|
||||||
if segsByBytes == 0 {
|
|
||||||
for _, b := range bufs {
|
|
||||||
if werr := u.WriteTo(b, addr); werr != nil {
|
|
||||||
return werr
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
maxChunk := maxGSOSegments
|
|
||||||
if segsByBytes < maxChunk {
|
|
||||||
maxChunk = segsByBytes
|
|
||||||
}
|
|
||||||
|
|
||||||
i := 0
|
|
||||||
for i < len(bufs) {
|
|
||||||
chunk := len(bufs) - i
|
|
||||||
if chunk > maxChunk {
|
|
||||||
chunk = maxChunk
|
|
||||||
}
|
|
||||||
for k := 0; k < chunk; k++ {
|
|
||||||
b := bufs[i+k]
|
|
||||||
if len(b) == 0 {
|
|
||||||
u.gsoIovs[k].Base = nil
|
|
||||||
setIovLen(&u.gsoIovs[k], 0)
|
|
||||||
} else {
|
|
||||||
u.gsoIovs[k].Base = &b[0]
|
|
||||||
setIovLen(&u.gsoIovs[k], len(b))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
setMsgIovlen(&u.gsoMsg, chunk)
|
|
||||||
binary.NativeEndian.PutUint16(u.gsoCmsg[unix.CmsgLen(0):unix.CmsgLen(0)+2], uint16(segSize))
|
|
||||||
|
|
||||||
if serr := u.sendmsgGSO(); serr != nil {
|
|
||||||
// Fall back to a per-packet loop for the remainder of the
|
|
||||||
// batch. Dropping the GSO call entirely is safer than
|
|
||||||
// returning mid-superpacket and losing bytes.
|
|
||||||
for k := 0; k < chunk; k++ {
|
|
||||||
if werr := u.WriteTo(bufs[i+k], addr); werr != nil {
|
|
||||||
return werr
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
i += chunk
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// sendmsgRawWriteGSO is the preallocated rawConn.Write callback for the GSO
|
|
||||||
// path. Reads the prebuilt u.gsoMsg and writes u.gsoSent / u.gsoErrno.
|
|
||||||
func (u *StdConn) sendmsgRawWriteGSO(fd uintptr) bool {
|
|
||||||
r1, _, errno := unix.Syscall(
|
|
||||||
unix.SYS_SENDMSG,
|
|
||||||
fd,
|
|
||||||
uintptr(unsafe.Pointer(&u.gsoMsg)),
|
|
||||||
0,
|
|
||||||
)
|
|
||||||
if errno == syscall.EAGAIN || errno == syscall.EWOULDBLOCK {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
u.gsoSent = int(r1)
|
|
||||||
u.gsoErrno = errno
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *StdConn) sendmsgGSO() error {
|
|
||||||
u.gsoSent = 0
|
|
||||||
u.gsoErrno = 0
|
|
||||||
if err := u.rawConn.Write(u.gsoFunc); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
if u.gsoErrno != 0 {
|
|
||||||
return &net.OpError{Op: "sendmsg", Err: u.gsoErrno}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *StdConn) sendmmsg(n int) (int, error) {
|
|
||||||
u.writeChunk = n
|
|
||||||
u.writeSent = 0
|
|
||||||
u.writeErrno = 0
|
|
||||||
if err := u.rawConn.Write(u.writeFunc); err != nil {
|
|
||||||
return u.writeSent, err
|
|
||||||
}
|
|
||||||
if u.writeErrno != 0 {
|
|
||||||
return u.writeSent, &net.OpError{Op: "sendmmsg", Err: u.writeErrno}
|
|
||||||
}
|
|
||||||
return u.writeSent, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// writeSockaddr encodes addr into buf (which must be at least
|
|
||||||
// SizeofSockaddrInet6 bytes). Returns the number of bytes used. If isV4 is
|
|
||||||
// true and addr is not a v4 (or v4-in-v6) address, returns an error.
|
|
||||||
func writeSockaddr(buf []byte, addr netip.AddrPort, isV4 bool) (int, error) {
|
|
||||||
ap := addr.Addr().Unmap()
|
|
||||||
if isV4 {
|
|
||||||
if !ap.Is4() {
|
|
||||||
return 0, ErrInvalidIPv6RemoteForSocket
|
|
||||||
}
|
|
||||||
// struct sockaddr_in: { sa_family_t(2), in_port_t(2, BE), in_addr(4), zero(8) }
|
|
||||||
// sa_family is host endian.
|
|
||||||
binary.NativeEndian.PutUint16(buf[0:2], unix.AF_INET)
|
|
||||||
binary.BigEndian.PutUint16(buf[2:4], addr.Port())
|
|
||||||
ip4 := ap.As4()
|
|
||||||
copy(buf[4:8], ip4[:])
|
|
||||||
for j := 8; j < 16; j++ {
|
|
||||||
buf[j] = 0
|
|
||||||
}
|
|
||||||
return unix.SizeofSockaddrInet4, nil
|
|
||||||
}
|
|
||||||
// struct sockaddr_in6: { sa_family_t(2), in_port_t(2, BE), flowinfo(4), in6_addr(16), scope_id(4) }
|
|
||||||
binary.NativeEndian.PutUint16(buf[0:2], unix.AF_INET6)
|
|
||||||
binary.BigEndian.PutUint16(buf[2:4], addr.Port())
|
|
||||||
binary.NativeEndian.PutUint32(buf[4:8], 0)
|
|
||||||
ip6 := addr.Addr().As16()
|
|
||||||
copy(buf[8:24], ip6[:])
|
|
||||||
binary.NativeEndian.PutUint32(buf[24:28], 0)
|
|
||||||
return unix.SizeofSockaddrInet6, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *StdConn) ReloadConfig(c *config.C) {
|
func (u *StdConn) ReloadConfig(c *config.C) {
|
||||||
b := c.GetInt("listen.read_buffer", 0)
|
b := c.GetInt("listen.read_buffer", 0)
|
||||||
if b > 0 {
|
if b > 0 {
|
||||||
|
|||||||
@@ -52,35 +52,3 @@ func (u *StdConn) PrepareRawMessages(n int) ([]rawMessage, [][]byte, [][]byte) {
|
|||||||
|
|
||||||
return msgs, buffers, names
|
return msgs, buffers, names
|
||||||
}
|
}
|
||||||
|
|
||||||
// prepareWriteMessages allocates one Mmsghdr/iovec/sockaddr scratch per slot,
|
|
||||||
// wired up so each writeMsgs[i] already points at writeIovs[i] and
|
|
||||||
// writeNames[i]. Callers fill in the iovec Base/Len, the sockaddr bytes, and
|
|
||||||
// Namelen before each sendmmsg.
|
|
||||||
func (u *StdConn) prepareWriteMessages(n int) {
|
|
||||||
u.writeMsgs = make([]rawMessage, n)
|
|
||||||
u.writeIovs = make([]iovec, n)
|
|
||||||
u.writeNames = make([][]byte, n)
|
|
||||||
for i := range u.writeMsgs {
|
|
||||||
u.writeNames[i] = make([]byte, unix.SizeofSockaddrInet6)
|
|
||||||
u.writeMsgs[i].Hdr.Iov = &u.writeIovs[i]
|
|
||||||
u.writeMsgs[i].Hdr.Iovlen = 1
|
|
||||||
u.writeMsgs[i].Hdr.Name = &u.writeNames[i][0]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func setIovLen(v *iovec, n int) {
|
|
||||||
v.Len = uint32(n)
|
|
||||||
}
|
|
||||||
|
|
||||||
func setMsgIovlen(m *msghdr, n int) {
|
|
||||||
m.Iovlen = uint32(n)
|
|
||||||
}
|
|
||||||
|
|
||||||
func setMsgControllen(m *msghdr, n int) {
|
|
||||||
m.Controllen = uint32(n)
|
|
||||||
}
|
|
||||||
|
|
||||||
func setCmsgLen(h *unix.Cmsghdr, n int) {
|
|
||||||
h.Len = uint32(n)
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -55,35 +55,3 @@ func (u *StdConn) PrepareRawMessages(n int) ([]rawMessage, [][]byte, [][]byte) {
|
|||||||
|
|
||||||
return msgs, buffers, names
|
return msgs, buffers, names
|
||||||
}
|
}
|
||||||
|
|
||||||
// prepareWriteMessages allocates one Mmsghdr/iovec/sockaddr scratch per slot,
|
|
||||||
// wired up so each writeMsgs[i] already points at writeIovs[i] and
|
|
||||||
// writeNames[i]. Callers fill in the iovec Base/Len, the sockaddr bytes, and
|
|
||||||
// Namelen before each sendmmsg.
|
|
||||||
func (u *StdConn) prepareWriteMessages(n int) {
|
|
||||||
u.writeMsgs = make([]rawMessage, n)
|
|
||||||
u.writeIovs = make([]iovec, n)
|
|
||||||
u.writeNames = make([][]byte, n)
|
|
||||||
for i := range u.writeMsgs {
|
|
||||||
u.writeNames[i] = make([]byte, unix.SizeofSockaddrInet6)
|
|
||||||
u.writeMsgs[i].Hdr.Iov = &u.writeIovs[i]
|
|
||||||
u.writeMsgs[i].Hdr.Iovlen = 1
|
|
||||||
u.writeMsgs[i].Hdr.Name = &u.writeNames[i][0]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func setIovLen(v *iovec, n int) {
|
|
||||||
v.Len = uint64(n)
|
|
||||||
}
|
|
||||||
|
|
||||||
func setMsgIovlen(m *msghdr, n int) {
|
|
||||||
m.Iovlen = uint64(n)
|
|
||||||
}
|
|
||||||
|
|
||||||
func setMsgControllen(m *msghdr, n int) {
|
|
||||||
m.Controllen = uint64(n)
|
|
||||||
}
|
|
||||||
|
|
||||||
func setCmsgLen(h *unix.Cmsghdr, n int) {
|
|
||||||
h.Len = uint64(n)
|
|
||||||
}
|
|
||||||
|
|||||||
+1
-22
@@ -140,7 +140,7 @@ func (u *RIOConn) bind(l *logrus.Logger, sa windows.Sockaddr) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *RIOConn) ListenOut(r EncReader, flush func()) error {
|
func (u *RIOConn) ListenOut(r EncReader) error {
|
||||||
buffer := make([]byte, MTU)
|
buffer := make([]byte, MTU)
|
||||||
|
|
||||||
var lastRecvErr time.Time
|
var lastRecvErr time.Time
|
||||||
@@ -162,7 +162,6 @@ func (u *RIOConn) ListenOut(r EncReader, flush func()) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
r(netip.AddrPortFrom(netip.AddrFrom16(rua.Addr).Unmap(), (rua.Port>>8)|((rua.Port&0xff)<<8)), buffer[:n])
|
r(netip.AddrPortFrom(netip.AddrFrom16(rua.Addr).Unmap(), (rua.Port>>8)|((rua.Port&0xff)<<8)), buffer[:n])
|
||||||
flush()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -317,26 +316,6 @@ func (u *RIOConn) WriteTo(buf []byte, ip netip.AddrPort) error {
|
|||||||
return winrio.SendEx(u.rq, dataBuffer, 1, nil, addressBuffer, nil, nil, 0, 0)
|
return winrio.SendEx(u.rq, dataBuffer, 1, nil, addressBuffer, nil, nil, 0, 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *RIOConn) WriteBatch(bufs [][]byte, addrs []netip.AddrPort) error {
|
|
||||||
for i, b := range bufs {
|
|
||||||
if err := u.WriteTo(b, addrs[i]); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *RIOConn) WriteSegmented(bufs [][]byte, addr netip.AddrPort, _ int) error {
|
|
||||||
for _, b := range bufs {
|
|
||||||
if err := u.WriteTo(b, addr); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *RIOConn) SupportsGSO() bool { return false }
|
|
||||||
|
|
||||||
func (u *RIOConn) LocalAddr() (netip.AddrPort, error) {
|
func (u *RIOConn) LocalAddr() (netip.AddrPort, error) {
|
||||||
sa, err := windows.Getsockname(u.sock)
|
sa, err := windows.Getsockname(u.sock)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
+1
-22
@@ -107,34 +107,13 @@ func (u *TesterConn) WriteTo(b []byte, addr netip.AddrPort) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (u *TesterConn) WriteBatch(bufs [][]byte, addrs []netip.AddrPort) error {
|
func (u *TesterConn) ListenOut(r EncReader) error {
|
||||||
for i, b := range bufs {
|
|
||||||
if err := u.WriteTo(b, addrs[i]); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *TesterConn) WriteSegmented(bufs [][]byte, addr netip.AddrPort, _ int) error {
|
|
||||||
for _, b := range bufs {
|
|
||||||
if err := u.WriteTo(b, addr); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (u *TesterConn) SupportsGSO() bool { return false }
|
|
||||||
|
|
||||||
func (u *TesterConn) ListenOut(r EncReader, flush func()) error {
|
|
||||||
for {
|
for {
|
||||||
p, ok := <-u.RxPackets
|
p, ok := <-u.RxPackets
|
||||||
if !ok {
|
if !ok {
|
||||||
return os.ErrClosed
|
return os.ErrClosed
|
||||||
}
|
}
|
||||||
r(p.From, p.Data)
|
r(p.From, p.Data)
|
||||||
flush()
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user