mirror of
https://github.com/slackhq/nebula.git
synced 2026-10-01 02:06:38 +02:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2df43dc218 | ||
|
|
72bf111209 |
@@ -43,15 +43,8 @@ runs:
|
||||
with:
|
||||
role-to-assume: ${{ inputs.role }}
|
||||
aws-region: ${{ inputs.region }}
|
||||
# An STS secret key with special characters does not survive the
|
||||
# pwsh -> make -> MSYS sh -> aws.exe chain, and SigV4 then signs with a
|
||||
# key that no longer matches, so the first S3 upload fails with
|
||||
# SignatureDoesNotMatch. Retries the assume until it comes back clean.
|
||||
# Same fix as DefinedNet/dnclient#867.
|
||||
special-characters-workaround: true
|
||||
# Overridden by the workaround above and kept for whenever that goes:
|
||||
# the default 12 rides out IAM trust-policy propagation, and once the
|
||||
# role is stable a real misconfiguration should fail fast.
|
||||
# Default is 12 retries to ride out IAM trust-policy propagation; once
|
||||
# the role is stable we want a real misconfiguration to fail fast.
|
||||
retry-max-attempts: 5
|
||||
|
||||
- name: Sign .exe files
|
||||
|
||||
@@ -73,11 +73,8 @@ jobs:
|
||||
build-darwin:
|
||||
name: Build Universal Darwin
|
||||
env:
|
||||
HAS_SIGNING_CREDS: ${{ secrets.APPLE_SIGNING_ROLE_ARN != '' }}
|
||||
HAS_SIGNING_CREDS: ${{ secrets.AC_USERNAME != '' }}
|
||||
runs-on: macos-latest
|
||||
permissions:
|
||||
id-token: write
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
@@ -86,68 +83,17 @@ jobs:
|
||||
go-version: '1.26'
|
||||
check-latest: true
|
||||
|
||||
# GitHub holds ARNs, not credentials, and ARNs outlive a rotation
|
||||
- name: Configure AWS credentials
|
||||
if: env.HAS_SIGNING_CREDS == 'true'
|
||||
uses: aws-actions/configure-aws-credentials@v6
|
||||
with:
|
||||
role-to-assume: ${{ secrets.APPLE_SIGNING_ROLE_ARN }}
|
||||
aws-region: us-east-2
|
||||
|
||||
# parse-json-secrets unpacks into SIGNING_* and ASC_*, masked on the way in
|
||||
- name: Fetch signing credentials
|
||||
if: env.HAS_SIGNING_CREDS == 'true'
|
||||
uses: aws-actions/aws-secretsmanager-get-secrets@v3
|
||||
with:
|
||||
parse-json-secrets: true
|
||||
secret-ids: |
|
||||
SIGNING,${{ secrets.APPLE_SIGNING_DEVELOPER_ID_ARN }}
|
||||
ASC,${{ secrets.APPLE_NOTARY_KEY_ARN }}
|
||||
|
||||
- name: Import certificates
|
||||
if: env.HAS_SIGNING_CREDS == 'true'
|
||||
uses: Apple-Actions/import-codesign-certs@v7
|
||||
with:
|
||||
p12-file-base64: ${{ env.SIGNING_P12_BASE64 }}
|
||||
p12-password: ${{ env.SIGNING_PASSWORD }}
|
||||
|
||||
# The action imports but does not check the chain validates, which is how a p12
|
||||
# missing its intermediate reaches a failing codesign
|
||||
- name: Check the identity is usable
|
||||
if: env.HAS_SIGNING_CREDS == 'true'
|
||||
run: |
|
||||
: "${SIGNING_IDENTITY_SHA1:?empty, so the secret has no identity_sha1}"
|
||||
identities=$(security find-identity -v -p codesigning signing_temp.keychain)
|
||||
case "$identities" in
|
||||
*"$SIGNING_IDENTITY_SHA1"*) ;;
|
||||
*) printf '%s\n' "$identities" >&2; exit 1 ;;
|
||||
esac
|
||||
|
||||
# notarytool wants the key as a file
|
||||
- name: Write the App Store Connect key
|
||||
if: env.HAS_SIGNING_CREDS == 'true'
|
||||
run: |
|
||||
mkdir -p ~/private_keys
|
||||
chmod 700 ~/private_keys
|
||||
key_path="$HOME/private_keys/AuthKey_${ASC_KEY_ID}.p8"
|
||||
(umask 077; printf '%s\n' "$ASC_PRIVATE_KEY" > "$key_path")
|
||||
echo "ASC_P8=$key_path" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Drop the credentials from the environment
|
||||
if: env.HAS_SIGNING_CREDS == 'true'
|
||||
run: |
|
||||
# The action's own inventory, so a new field in a secret is covered
|
||||
python3 -c '
|
||||
import json, os
|
||||
raw = os.environ.get("SECRETS_LIST_CLEAN_UP")
|
||||
if raw is None and os.environ.get("SIGNING_P12_BASE64"):
|
||||
raise SystemExit("SECRETS_LIST_CLEAN_UP is gone, fetched secrets are not being scrubbed")
|
||||
keep = {"SIGNING_IDENTITY_SHA1", "ASC_KEY_ID", "ASC_ISSUER_ID"}
|
||||
names = [n for n in json.loads(raw or "[]") if n not in keep]
|
||||
print("\n".join(f"{n}=" for n in dict.fromkeys(names)))
|
||||
' >> "$GITHUB_ENV"
|
||||
p12-file-base64: ${{ secrets.APPLE_DEVELOPER_CERTIFICATE_P12_BASE64 }}
|
||||
p12-password: ${{ secrets.APPLE_DEVELOPER_CERTIFICATE_PASSWORD }}
|
||||
|
||||
- name: Build, sign, and notarize
|
||||
env:
|
||||
AC_USERNAME: ${{ secrets.AC_USERNAME }}
|
||||
AC_PASSWORD: ${{ secrets.AC_PASSWORD }}
|
||||
run: |
|
||||
rm -rf release
|
||||
mkdir release
|
||||
@@ -156,34 +102,17 @@ jobs:
|
||||
lipo -create -output ./release/nebula ./build/darwin-amd64/nebula ./build/darwin-arm64/nebula
|
||||
lipo -create -output ./release/nebula-cert ./build/darwin-amd64/nebula-cert ./build/darwin-arm64/nebula-cert
|
||||
|
||||
# Unset in a fork, which has no credentials to sign with
|
||||
if [ -n "$SIGNING_IDENTITY_SHA1" ]; then
|
||||
codesign -s "$SIGNING_IDENTITY_SHA1" -f -v --timestamp --options=runtime -i "net.defined.nebula" ./release/nebula
|
||||
codesign -s "$SIGNING_IDENTITY_SHA1" -f -v --timestamp --options=runtime -i "net.defined.nebula-cert" ./release/nebula-cert
|
||||
if [ -n "$AC_USERNAME" ]; then
|
||||
codesign -s "10BC1FDDEB6CE753550156C0669109FAC49E4D1E" -f -v --timestamp --options=runtime -i "net.defined.nebula" ./release/nebula
|
||||
codesign -s "10BC1FDDEB6CE753550156C0669109FAC49E4D1E" -f -v --timestamp --options=runtime -i "net.defined.nebula-cert" ./release/nebula-cert
|
||||
fi
|
||||
|
||||
zip -j release/nebula-darwin.zip release/nebula-cert release/nebula
|
||||
|
||||
if [ -n "$ASC_P8" ]; then
|
||||
xcrun notarytool submit ./release/nebula-darwin.zip --key "$ASC_P8" --key-id "$ASC_KEY_ID" --issuer "$ASC_ISSUER_ID" --wait
|
||||
if [ -n "$AC_USERNAME" ]; then
|
||||
xcrun notarytool submit ./release/nebula-darwin.zip --team-id "576H3XS7FP" --apple-id "$AC_USERNAME" --password "$AC_PASSWORD" --wait
|
||||
fi
|
||||
|
||||
- name: Drop the signing key
|
||||
if: always() && env.HAS_SIGNING_CREDS == 'true'
|
||||
run: |
|
||||
# Locked, not deleted: import-codesign-certs deletes it in its own post
|
||||
# step and fails the job if it is already gone. Locked is unusable.
|
||||
security lock-keychain signing_temp.keychain || true
|
||||
rm -f "$ASC_P8"
|
||||
# Nothing later in this job needs AWS
|
||||
python3 -c '
|
||||
import json, os
|
||||
names = json.loads(os.environ.get("SECRETS_LIST_CLEAN_UP") or "[]")
|
||||
names += ["ASC_P8", "SIGNING_IDENTITY_SHA1", "ASC_KEY_ID", "ASC_ISSUER_ID",
|
||||
"AWS_ACCESS_KEY_ID", "AWS_SECRET_ACCESS_KEY", "AWS_SESSION_TOKEN"]
|
||||
print("\n".join(f"{n}=" for n in dict.fromkeys(names)))
|
||||
' >> "$GITHUB_ENV"
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
|
||||
+1
-28
@@ -7,31 +7,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [1.11.1] - 2026-08-21
|
||||
|
||||
See the [v1.11.1](https://github.com/slackhq/nebula/milestone/30?closed=1) milestone for a complete list of changes.
|
||||
|
||||
### Changed
|
||||
|
||||
- IPv6 packets whose next header is a protocol Nebula does not parse (SCTP, GRE, IP-in-IP, etc.) are now
|
||||
classified as that protocol with no ports, closing a firewall bypass where a crafted payload could steer
|
||||
the classifier into reading one as TCP/UDP and matching a TCP/UDP rule. These packets are now matched as
|
||||
their true protocol, so only a `proto: any` rule allows them. If you carry one of these protocols over the
|
||||
overlay, confirm a `proto: any` rule covers it before upgrading, it may have been passing only through this
|
||||
bypass. (#1840)
|
||||
- Drop the dependency on `github.com/cyberdelia/go-metrics-graphite`, which has been unmaintained for over ten
|
||||
years, by inlining the small amount of code Nebula used. (#1832)
|
||||
|
||||
### Fixed
|
||||
|
||||
- The ICMPv6 type was read from the wrong byte when classifying IPv6 packets, so the echo identifier used
|
||||
for conntrack was never picked up. (#1840)
|
||||
- Enforce outbound message counter limits so a tunnel is rehandshaked before the counter can wrap, preventing
|
||||
nonce reuse. This is unreachable in practice, but is enforced as a defense-in-depth measure. (#1841)
|
||||
- Prevent `nebula-cert ca` from running out of memory on 32bit systems when generating encrypted private keys. (#1834)
|
||||
- Tolerate `ErrDumpInterrupted` when listing tun addresses on Linux, so a transient interrupted netlink dump
|
||||
no longer aborts startup. (#1835)
|
||||
|
||||
## [1.11.0] - 2026-07-23
|
||||
|
||||
See the [v1.11.0](https://github.com/slackhq/nebula/milestone/25?closed=1) milestone for a complete list of changes.
|
||||
@@ -895,9 +870,7 @@ created.)
|
||||
|
||||
- Initial public release.
|
||||
|
||||
[Unreleased]: https://github.com/slackhq/nebula/compare/v1.11.1...HEAD
|
||||
[1.11.1]: https://github.com/slackhq/nebula/releases/tag/v1.11.1
|
||||
[1.11.0]: https://github.com/slackhq/nebula/releases/tag/v1.11.0
|
||||
[Unreleased]: https://github.com/slackhq/nebula/compare/v1.10.3...HEAD
|
||||
[1.10.3]: https://github.com/slackhq/nebula/releases/tag/v1.10.3
|
||||
[1.10.2]: https://github.com/slackhq/nebula/releases/tag/v1.10.2
|
||||
[1.10.1]: https://github.com/slackhq/nebula/releases/tag/v1.10.1
|
||||
|
||||
+2
-17
@@ -8,7 +8,6 @@ import (
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
"math/bits"
|
||||
"net/netip"
|
||||
"os"
|
||||
"strings"
|
||||
@@ -45,20 +44,6 @@ type caFlags struct {
|
||||
}
|
||||
|
||||
func newCaFlags() *caFlags {
|
||||
// prevent running out of memory on 32-bit systems by defaulting to
|
||||
// RFC9106's recommendation for memory-constrained environments
|
||||
var (
|
||||
defaultArgonMemory uint
|
||||
defaultArgonIterations uint
|
||||
)
|
||||
if bits.UintSize == 32 {
|
||||
defaultArgonMemory = 64 * 1024
|
||||
defaultArgonIterations = 3
|
||||
} else {
|
||||
defaultArgonMemory = 2 * 1024 * 1024
|
||||
defaultArgonIterations = 1
|
||||
}
|
||||
|
||||
cf := caFlags{set: flag.NewFlagSet("ca", flag.ContinueOnError)}
|
||||
cf.set.Usage = func() {}
|
||||
cf.name = cf.set.String("name", "", "Required: name of the certificate authority")
|
||||
@@ -70,9 +55,9 @@ func newCaFlags() *caFlags {
|
||||
cf.groups = cf.set.String("groups", "", "Optional: comma separated list of groups. This will limit which groups subordinate certs can use")
|
||||
cf.networks = cf.set.String("networks", "", "Optional: comma separated list of ip address and network in CIDR notation. This will limit which ip addresses and networks subordinate certs can use in networks")
|
||||
cf.unsafeNetworks = cf.set.String("unsafe-networks", "", "Optional: comma separated list of ip address and network in CIDR notation. This will limit which ip addresses and networks subordinate certs can use in unsafe networks")
|
||||
cf.argonMemory = cf.set.Uint("argon-memory", defaultArgonMemory, "Optional: Argon2 memory parameter (in KiB) used for encrypted private key passphrase")
|
||||
cf.argonMemory = cf.set.Uint("argon-memory", 2*1024*1024, "Optional: Argon2 memory parameter (in KiB) used for encrypted private key passphrase")
|
||||
cf.argonParallelism = cf.set.Uint("argon-parallelism", 4, "Optional: Argon2 parallelism parameter used for encrypted private key passphrase")
|
||||
cf.argonIterations = cf.set.Uint("argon-iterations", defaultArgonIterations, "Optional: Argon2 iterations parameter used for encrypted private key passphrase")
|
||||
cf.argonIterations = cf.set.Uint("argon-iterations", 1, "Optional: Argon2 iterations parameter used for encrypted private key passphrase")
|
||||
cf.encryption = cf.set.Bool("encrypt", false, "Optional: prompt for passphrase and write out-key in an encrypted format")
|
||||
cf.curve = cf.set.String("curve", "25519", "EdDSA/ECDSA Curve (25519, P256)")
|
||||
cf.p11url = p11Flag(cf.set)
|
||||
|
||||
@@ -7,9 +7,7 @@ import (
|
||||
"bytes"
|
||||
"encoding/pem"
|
||||
"errors"
|
||||
"math/bits"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -24,18 +22,6 @@ func Test_caSummary(t *testing.T) {
|
||||
}
|
||||
|
||||
func Test_caHelp(t *testing.T) {
|
||||
var (
|
||||
defaultArgonMemory string
|
||||
defaultArgonIterations string
|
||||
)
|
||||
if bits.UintSize == 32 {
|
||||
defaultArgonMemory = strconv.Itoa(64 * 1024)
|
||||
defaultArgonIterations = strconv.Itoa(3)
|
||||
} else {
|
||||
defaultArgonMemory = strconv.Itoa(2 * 1024 * 1024)
|
||||
defaultArgonIterations = strconv.Itoa(1)
|
||||
}
|
||||
|
||||
ob := &bytes.Buffer{}
|
||||
caHelp(ob)
|
||||
assert.Equal(
|
||||
@@ -43,9 +29,9 @@ func Test_caHelp(t *testing.T) {
|
||||
"Usage of "+os.Args[0]+" ca <flags>: create a self signed certificate authority\n"+
|
||||
" Pass \"-\" to any path flag to read from stdin or write to stdout.\n"+
|
||||
" -argon-iterations uint\n"+
|
||||
" \tOptional: Argon2 iterations parameter used for encrypted private key passphrase (default "+defaultArgonIterations+")\n"+
|
||||
" \tOptional: Argon2 iterations parameter used for encrypted private key passphrase (default 1)\n"+
|
||||
" -argon-memory uint\n"+
|
||||
" \tOptional: Argon2 memory parameter (in KiB) used for encrypted private key passphrase (default "+defaultArgonMemory+")\n"+
|
||||
" \tOptional: Argon2 memory parameter (in KiB) used for encrypted private key passphrase (default 2097152)\n"+
|
||||
" -argon-parallelism uint\n"+
|
||||
" \tOptional: Argon2 parallelism parameter used for encrypted private key passphrase (default 4)\n"+
|
||||
" -curve string\n"+
|
||||
@@ -202,16 +188,10 @@ func Test_ca(t *testing.T) {
|
||||
k, _ := pem.Decode(rb)
|
||||
ned, err := cert.UnmarshalNebulaEncryptedData(k.Bytes)
|
||||
require.NoError(t, err)
|
||||
|
||||
if bits.UintSize == 32 {
|
||||
assert.Equal(t, uint32(64*1024), ned.EncryptionMetadata.Argon2Parameters.Memory)
|
||||
assert.Equal(t, uint32(3), ned.EncryptionMetadata.Argon2Parameters.Iterations)
|
||||
} else {
|
||||
assert.Equal(t, uint32(2*1024*1024), ned.EncryptionMetadata.Argon2Parameters.Memory)
|
||||
assert.Equal(t, uint32(1), ned.EncryptionMetadata.Argon2Parameters.Iterations)
|
||||
}
|
||||
|
||||
// we won't know salt in advance, so just check start of string
|
||||
assert.Equal(t, uint32(2*1024*1024), ned.EncryptionMetadata.Argon2Parameters.Memory)
|
||||
assert.Equal(t, uint8(4), ned.EncryptionMetadata.Argon2Parameters.Parallelism)
|
||||
assert.Equal(t, uint32(1), ned.EncryptionMetadata.Argon2Parameters.Iterations)
|
||||
|
||||
// verify the key is valid and decrypt-able
|
||||
var curve cert.Curve
|
||||
|
||||
@@ -323,12 +323,6 @@ func (cm *connectionManager) makeTrafficDecision(localIndex uint32, now time.Tim
|
||||
return closeTunnel, hostinfo, nil
|
||||
}
|
||||
|
||||
if hostinfo.ConnectionState != nil && hostinfo.ConnectionState.messageCounter.Load() >= RejectAfterMessages {
|
||||
// Send path can't encrypt a CloseTunnel notify, so just delete locally; the peer recovers via recv_error.
|
||||
hostinfo.logger(cm.l).Error("Dropping tunnel, message counter is exhausted")
|
||||
return deleteTunnel, hostinfo, nil
|
||||
}
|
||||
|
||||
primary := cm.hostMap.Hosts[hostinfo.vpnAddrs[0]]
|
||||
mainHostInfo := true
|
||||
if primary != nil && primary != hostinfo {
|
||||
@@ -454,11 +448,6 @@ func (cm *connectionManager) shouldSwapPrimary(current *HostInfo) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
if current.ConnectionState.messageCounter.Load() >= RehandshakeAfterMessages {
|
||||
// This tunnel is being rolled for counter exhaustion, never swap back onto its spent key.
|
||||
return false
|
||||
}
|
||||
|
||||
crt := cm.intf.pki.getCertState().getCertificate(current.ConnectionState.myCert.Version())
|
||||
if crt == nil {
|
||||
//my cert was reloaded away. We should definitely swap from this tunnel
|
||||
@@ -555,15 +544,6 @@ func (cm *connectionManager) tryRehandshake(hostinfo *HostInfo) {
|
||||
"reason", "current cert version < pki.initiatingVersion",
|
||||
)
|
||||
|
||||
cm.intf.handshakeManager.StartHandshake(hostinfo.vpnAddrs[0], nil)
|
||||
return
|
||||
}
|
||||
if hostinfo.ConnectionState.messageCounter.Load() >= RehandshakeAfterMessages {
|
||||
cm.l.Info("Re-handshaking with remote",
|
||||
"vpnAddrs", hostinfo.vpnAddrs,
|
||||
"reason", "message counter rehandshake threshold reached",
|
||||
)
|
||||
|
||||
cm.intf.handshakeManager.StartHandshake(hostinfo.vpnAddrs[0], nil)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -199,79 +199,6 @@ func Test_NewConnectionManagerTest2(t *testing.T) {
|
||||
assert.Contains(t, nc.hostMap.Hosts, hostinfo.vpnAddrs[0])
|
||||
}
|
||||
|
||||
func Test_NewConnectionManager_CounterLimits(t *testing.T) {
|
||||
l := test.NewLogger()
|
||||
localrange := netip.MustParsePrefix("10.1.1.1/24")
|
||||
vpnIp := netip.MustParseAddr("172.1.1.2")
|
||||
preferredRanges := []netip.Prefix{localrange}
|
||||
|
||||
// Very incomplete mock objects
|
||||
hostMap := newHostMap(l)
|
||||
hostMap.preferredRanges.Store(&preferredRanges)
|
||||
|
||||
cs := &CertState{
|
||||
initiatingVersion: cert.Version1,
|
||||
privateKey: []byte{},
|
||||
v1Cert: &dummyCert{version: cert.Version1},
|
||||
v1Credential: nil,
|
||||
}
|
||||
|
||||
lh := newTestLighthouse()
|
||||
ifce := &Interface{
|
||||
hostMap: hostMap,
|
||||
inside: &overlaytest.NoopTun{},
|
||||
outside: &udp.NoopConn{},
|
||||
firewall: &Firewall{},
|
||||
lightHouse: lh,
|
||||
pki: &PKI{},
|
||||
myVpnAddrs: []netip.Addr{netip.MustParseAddr("172.1.1.1")}, // sorts below vpnIp so shouldSwapPrimary can proceed
|
||||
handshakeManager: NewHandshakeManager(l, hostMap, lh, &udp.NoopConn{}, defaultHandshakeConfig),
|
||||
l: l,
|
||||
}
|
||||
ifce.pki.cs.Store(cs)
|
||||
|
||||
conf := config.NewC(test.NewLogger())
|
||||
punchy := NewPunchyFromConfig(test.NewLogger(), conf, nil)
|
||||
nc := newConnectionManagerFromConfig(test.NewLogger(), conf, hostMap, punchy)
|
||||
nc.intf = ifce
|
||||
|
||||
hostinfo := &HostInfo{
|
||||
vpnAddrs: []netip.Addr{vpnIp},
|
||||
localIndexId: 1099,
|
||||
remoteIndexId: 9901,
|
||||
}
|
||||
hostinfo.ConnectionState = &ConnectionState{
|
||||
myCert: &dummyCert{version: cert.Version1},
|
||||
}
|
||||
nc.hostMap.unlockedAddHostInfo(hostinfo, ifce)
|
||||
|
||||
// Below the rehandshake threshold, no handshake is started
|
||||
hostinfo.ConnectionState.messageCounter.Store(RehandshakeAfterMessages - 1)
|
||||
nc.tryRehandshake(hostinfo)
|
||||
assert.Nil(t, ifce.handshakeManager.QueryVpnAddr(vpnIp))
|
||||
|
||||
// A tunnel on its current cert would normally swap to primary
|
||||
assert.True(t, nc.shouldSwapPrimary(hostinfo))
|
||||
|
||||
// At the rehandshake threshold, a new handshake is started
|
||||
hostinfo.ConnectionState.messageCounter.Store(RehandshakeAfterMessages)
|
||||
nc.tryRehandshake(hostinfo)
|
||||
assert.NotNil(t, ifce.handshakeManager.QueryVpnAddr(vpnIp))
|
||||
|
||||
// An exhausted tunnel being rolled must never swap back to primary onto its spent key
|
||||
assert.False(t, nc.shouldSwapPrimary(hostinfo))
|
||||
|
||||
// Still below the reject limit, the tunnel stays up
|
||||
nc.In(hostinfo)
|
||||
decision, _, _ := nc.makeTrafficDecision(hostinfo.localIndexId, time.Now())
|
||||
assert.Equal(t, tryRehandshake, decision)
|
||||
|
||||
// At the reject limit, the tunnel is deleted locally without a doomed CloseTunnel notify
|
||||
hostinfo.ConnectionState.messageCounter.Store(RejectAfterMessages)
|
||||
decision, _, _ = nc.makeTrafficDecision(hostinfo.localIndexId, time.Now())
|
||||
assert.Equal(t, deleteTunnel, decision)
|
||||
}
|
||||
|
||||
func Test_NewConnectionManager_DisconnectInactive(t *testing.T) {
|
||||
l := test.NewLogger()
|
||||
localrange := netip.MustParsePrefix("10.1.1.1/24")
|
||||
|
||||
+3
-30
@@ -2,7 +2,6 @@ package nebula
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
@@ -13,18 +12,7 @@ import (
|
||||
"github.com/slackhq/nebula/noiseutil"
|
||||
)
|
||||
|
||||
const (
|
||||
ReplayWindow = 1024
|
||||
|
||||
// RehandshakeAfterMessages rolls keys inside the AES-GCM data-volume margin (~2^-36 advantage at 64KB frames).
|
||||
RehandshakeAfterMessages = uint64(1) << 34
|
||||
|
||||
// RejectAfterMessages is the nonce ceiling enforced by noiseutil; a tunnel here is deleted locally, not notified.
|
||||
RejectAfterMessages = noiseutil.RejectAfterMessages
|
||||
)
|
||||
|
||||
// RehandshakeAfterMessages must stay below RejectAfterMessages so tunnels roll before the hard send stop.
|
||||
const _ = RejectAfterMessages - RehandshakeAfterMessages
|
||||
const ReplayWindow = 1024
|
||||
|
||||
type ConnectionState struct {
|
||||
eKey noiseutil.CipherState
|
||||
@@ -42,12 +30,7 @@ type ConnectionState struct {
|
||||
// completed handshake.Result. It seeds messageCounter and the replay window so
|
||||
// that the post-handshake message indices already used on the wire don't count
|
||||
// as missed traffic in the data plane.
|
||||
func newConnectionStateFromResult(r *handshake.Result) (*ConnectionState, error) {
|
||||
// Refuse a MessageIndex too big for the replay window: it can only be a bug, and would spin the seed loop below.
|
||||
if r.MessageIndex >= ReplayWindow {
|
||||
return nil, fmt.Errorf("handshake message index %d exceeds replay window", r.MessageIndex)
|
||||
}
|
||||
|
||||
func newConnectionStateFromResult(r *handshake.Result) *ConnectionState {
|
||||
ci := &ConnectionState{
|
||||
myCert: r.MyCert,
|
||||
initiator: r.Initiator,
|
||||
@@ -60,7 +43,7 @@ func newConnectionStateFromResult(r *handshake.Result) (*ConnectionState, error)
|
||||
for i := uint64(1); i <= r.MessageIndex; i++ {
|
||||
ci.window.Update(nil, i)
|
||||
}
|
||||
return ci, nil
|
||||
return ci
|
||||
}
|
||||
|
||||
func (cs *ConnectionState) MarshalJSON() ([]byte, error) {
|
||||
@@ -71,16 +54,6 @@ func (cs *ConnectionState) MarshalJSON() ([]byte, error) {
|
||||
})
|
||||
}
|
||||
|
||||
// NextMessageCounter reserves the next 1-based counter; RejectAfterMessages is the first we refuse, pinned to not wrap.
|
||||
func (cs *ConnectionState) NextMessageCounter() (uint64, bool) {
|
||||
c := cs.messageCounter.Add(1)
|
||||
if c >= RejectAfterMessages {
|
||||
cs.messageCounter.Store(RejectAfterMessages)
|
||||
return c, false
|
||||
}
|
||||
return c, true
|
||||
}
|
||||
|
||||
func (cs *ConnectionState) Curve() cert.Curve {
|
||||
return cs.myCert.Curve()
|
||||
}
|
||||
|
||||
@@ -6,12 +6,10 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/flynn/noise"
|
||||
"github.com/rcrowley/go-metrics"
|
||||
"github.com/slackhq/nebula/cert"
|
||||
ct "github.com/slackhq/nebula/cert_test"
|
||||
"github.com/slackhq/nebula/handshake"
|
||||
"github.com/slackhq/nebula/header"
|
||||
"github.com/slackhq/nebula/test"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
@@ -81,51 +79,11 @@ func runTestHandshake(t *testing.T) (initR, respR *handshake.Result) {
|
||||
return initR, respR
|
||||
}
|
||||
|
||||
func TestConnectionState_NextMessageCounter(t *testing.T) {
|
||||
cs := &ConnectionState{}
|
||||
cs.messageCounter.Store(RejectAfterMessages - 2)
|
||||
|
||||
c, ok := cs.NextMessageCounter()
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, RejectAfterMessages-1, c)
|
||||
|
||||
// Hitting the limit refuses and pins the counter there
|
||||
c, ok = cs.NextMessageCounter()
|
||||
assert.False(t, ok)
|
||||
assert.Equal(t, RejectAfterMessages, c)
|
||||
assert.Equal(t, RejectAfterMessages, cs.messageCounter.Load())
|
||||
|
||||
// Continued send attempts stay refused and the counter never wraps
|
||||
for i := 0; i < 10; i++ {
|
||||
_, ok = cs.NextMessageCounter()
|
||||
assert.False(t, ok)
|
||||
}
|
||||
assert.Equal(t, RejectAfterMessages, cs.messageCounter.Load())
|
||||
}
|
||||
|
||||
// TestSendNoMetricsDropsExhausted drives the send path to the exhausted drop; metric and out flag prove it.
|
||||
func TestSendNoMetricsDropsExhausted(t *testing.T) {
|
||||
initR, _ := runTestHandshake(t)
|
||||
ci, err := newConnectionStateFromResult(initR)
|
||||
require.NoError(t, err)
|
||||
ci.messageCounter.Store(RejectAfterMessages - 1)
|
||||
|
||||
f := &Interface{l: test.NewLogger(), messageMetrics: &MessageMetrics{txExhausted: metrics.NewCounter()}}
|
||||
hostinfo := &HostInfo{vpnAddrs: []netip.Addr{netip.MustParseAddr("10.0.0.1")}, ConnectionState: ci}
|
||||
|
||||
f.sendNoMetrics(header.Message, 0, ci, hostinfo, netip.AddrPort{}, []byte{}, make([]byte, 12), make([]byte, mtu), 0)
|
||||
|
||||
// The crossing send is refused: it records an exhaustion drop and never reaches connectionManager.Out.
|
||||
assert.Equal(t, int64(1), f.messageMetrics.txExhausted.Count())
|
||||
assert.False(t, hostinfo.out.Load())
|
||||
}
|
||||
|
||||
func TestNewConnectionStateFromResult(t *testing.T) {
|
||||
initR, respR := runTestHandshake(t)
|
||||
|
||||
t.Run("initiator", func(t *testing.T) {
|
||||
ci, err := newConnectionStateFromResult(initR)
|
||||
require.NoError(t, err)
|
||||
ci := newConnectionStateFromResult(initR)
|
||||
assert.True(t, ci.initiator)
|
||||
assert.Equal(t, initR.MyCert, ci.myCert)
|
||||
assert.Equal(t, initR.RemoteCert, ci.peerCert)
|
||||
@@ -144,17 +102,8 @@ func TestNewConnectionStateFromResult(t *testing.T) {
|
||||
assert.True(t, ci.window.Check(nil, 3), "counter 3 must not be pre-seeded")
|
||||
})
|
||||
|
||||
t.Run("message index too large is refused", func(t *testing.T) {
|
||||
bad := *initR
|
||||
bad.MessageIndex = ReplayWindow
|
||||
ci, err := newConnectionStateFromResult(&bad)
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, ci)
|
||||
})
|
||||
|
||||
t.Run("responder", func(t *testing.T) {
|
||||
ci, err := newConnectionStateFromResult(respR)
|
||||
require.NoError(t, err)
|
||||
ci := newConnectionStateFromResult(respR)
|
||||
assert.False(t, ci.initiator)
|
||||
assert.Equal(t, respR.MyCert, ci.myCert)
|
||||
assert.Equal(t, respR.RemoteCert, ci.peerCert)
|
||||
|
||||
@@ -52,6 +52,7 @@ type Control struct {
|
||||
sshStart func()
|
||||
statsStart func()
|
||||
dnsStart func()
|
||||
infoAPIStart func()
|
||||
lighthouseStart func()
|
||||
networkChangeStart func(rebind func())
|
||||
connectionManagerStart func(context.Context)
|
||||
@@ -108,6 +109,9 @@ func (c *Control) Start() error {
|
||||
if c.networkChangeStart != nil {
|
||||
go c.networkChangeStart(c.RebindUDPServer)
|
||||
}
|
||||
if c.infoAPIStart != nil {
|
||||
go c.infoAPIStart()
|
||||
}
|
||||
if c.connectionManagerStart != nil {
|
||||
go c.connectionManagerStart(c.ctx)
|
||||
}
|
||||
|
||||
+3
-22
@@ -258,31 +258,12 @@ func (d *dnsServer) QueryCert(data string) string {
|
||||
return ""
|
||||
}
|
||||
|
||||
// The hostmap only ever contains peers we have handshaked with, so it never carries an entry for ourselves.
|
||||
// Answer self lookups straight from the local cert state.
|
||||
if cs := d.certState(); cs != nil && cs.myVpnAddrsTable != nil && cs.myVpnAddrsTable.Contains(ip) {
|
||||
c := cs.GetDefaultCertificate()
|
||||
if c == nil {
|
||||
return ""
|
||||
}
|
||||
b, err := c.MarshalJSON()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
hostinfo := d.hostMap.QueryVpnAddr(ip)
|
||||
if hostinfo == nil {
|
||||
crt := findCertificateForVpnAddr(d.certState(), d.hostMap, ip)
|
||||
if crt == nil {
|
||||
return ""
|
||||
}
|
||||
|
||||
q := hostinfo.GetCert()
|
||||
if q == nil {
|
||||
return ""
|
||||
}
|
||||
|
||||
b, err := q.Certificate.MarshalJSON()
|
||||
b, err := crt.MarshalJSON()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
|
||||
@@ -0,0 +1,136 @@
|
||||
//go:build e2e_testing
|
||||
// +build e2e_testing
|
||||
|
||||
package e2e
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/slackhq/nebula"
|
||||
"github.com/slackhq/nebula/cert"
|
||||
"github.com/slackhq/nebula/cert_test"
|
||||
"github.com/slackhq/nebula/e2e/router"
|
||||
"github.com/slackhq/nebula/udp"
|
||||
)
|
||||
|
||||
// TestRecoveryTiming measures how long a tunnel takes to come back after the peer stops accepting our traffic,
|
||||
// which is what a laptop waking on a new network looks like from the peer's side: its NAT has no state for where
|
||||
// we are now, so everything we send disappears.
|
||||
//
|
||||
// It is a measurement, not a pass/fail assertion. Recovery is timed to the moment the peer punches back at us,
|
||||
// since that is when its NAT opens and the tunnel is usable again.
|
||||
//
|
||||
// go test -tags e2e_testing -v -run TestRecoveryTiming ./e2e/
|
||||
func TestRecoveryTiming(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
rebind bool
|
||||
}{
|
||||
{"no trigger", false},
|
||||
{"rebind counter", true},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
d, lost := measureRecovery(t, tc.rebind)
|
||||
t.Logf("RESULT %-16s recovered in %-9v (%d packets lost)", tc.name, d.Round(time.Millisecond), lost)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// measureRecovery returns how long until the peer punched back, and how many of our packets died meanwhile. When
|
||||
// rebind is true we call RebindUDPServer once the tunnel goes dark, which is what the darwin network change
|
||||
// monitor does and what iOS has always done. When false, nothing tells nebula anything is wrong.
|
||||
func measureRecovery(t *testing.T, rebind bool) (time.Duration, int) {
|
||||
t.Helper()
|
||||
ca, _, caKey, _ := cert_test.NewTestCaCert(cert.Version2, cert.Curve_CURVE25519, time.Now(), time.Now().Add(10*time.Minute), nil, nil, []string{})
|
||||
|
||||
lhControl, lhVpnIpNet, lhUdpAddr, _ := newSimpleServer(cert.Version2, ca, caKey, "lh", "10.128.0.1/24", m{
|
||||
"lighthouse": m{"am_lighthouse": true},
|
||||
})
|
||||
|
||||
peerCfg := m{
|
||||
"lighthouse": m{
|
||||
"hosts": []any{lhVpnIpNet[0].Addr().String()},
|
||||
"interval": 600,
|
||||
"local_allow_list": m{
|
||||
"10.0.0.0/24": true,
|
||||
"::/0": false,
|
||||
},
|
||||
},
|
||||
"static_host_map": m{
|
||||
lhVpnIpNet[0].Addr().String(): []any{lhUdpAddr.String()},
|
||||
},
|
||||
}
|
||||
|
||||
myControl, myVpnIpNet, myUdpAddr, _ := newSimpleServer(cert.Version2, ca, caKey, "me", "10.128.0.2/24", peerCfg)
|
||||
theirControl, theirVpnIpNet, theirUdpAddr, _ := newSimpleServer(cert.Version2, ca, caKey, "them", "10.128.0.3/24", peerCfg)
|
||||
|
||||
r := router.NewR(t, lhControl, myControl, theirControl)
|
||||
defer r.RenderFlow()
|
||||
defer func() {
|
||||
lhControl.Stop()
|
||||
myControl.Stop()
|
||||
theirControl.Stop()
|
||||
}()
|
||||
|
||||
lhControl.Start()
|
||||
myControl.Start()
|
||||
theirControl.Start()
|
||||
r.RouteFor(time.Millisecond * 500)
|
||||
|
||||
myControl.InjectLightHouseAddr(theirVpnIpNet[0].Addr(), theirUdpAddr)
|
||||
theirControl.InjectLightHouseAddr(myVpnIpNet[0].Addr(), myUdpAddr)
|
||||
|
||||
myControl.InjectTunPacket(BuildTunUDPPacket(theirVpnIpNet[0].Addr(), 80, myVpnIpNet[0].Addr(), 80, []byte("establish")))
|
||||
r.RouteFor(time.Second)
|
||||
if myControl.GetHostInfoByVpnAddr(theirVpnIpNet[0].Addr(), false) == nil {
|
||||
t.Fatal("failed to establish the tunnel we are measuring")
|
||||
}
|
||||
r.RouteFor(time.Millisecond * 500)
|
||||
|
||||
// From here the peer's NAT has no state for us, everything we send it disappears
|
||||
start := time.Now()
|
||||
blackholed := 0
|
||||
var recovered time.Duration
|
||||
|
||||
if rebind {
|
||||
myControl.RebindUDPServer()
|
||||
}
|
||||
|
||||
// Keep the tun busy the way someone retrying a stalled connection would
|
||||
stop := make(chan struct{})
|
||||
defer close(stop)
|
||||
go func() {
|
||||
tick := time.NewTicker(time.Millisecond * 200)
|
||||
defer tick.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-stop:
|
||||
return
|
||||
case <-tick.C:
|
||||
myControl.InjectTunPacket(BuildTunUDPPacket(
|
||||
theirVpnIpNet[0].Addr(), 80, myVpnIpNet[0].Addr(), 80, []byte("retry")))
|
||||
}
|
||||
}
|
||||
}()
|
||||
|
||||
r.RouteForAllExitFuncOrTimeout(time.Second*30, func(p *udp.Packet, c *nebula.Control) router.ExitType {
|
||||
if c == theirControl && p.From == myControl.GetUDPAddr() {
|
||||
blackholed++
|
||||
return router.Drop
|
||||
}
|
||||
|
||||
// The peer reaching us directly is the moment its NAT opened, whether that is a punch or a handshake
|
||||
if c == myControl && p.From == theirUdpAddr {
|
||||
recovered = time.Since(start)
|
||||
return router.RouteAndExit
|
||||
}
|
||||
|
||||
return router.KeepRouting
|
||||
})
|
||||
|
||||
if recovered == 0 {
|
||||
t.Fatalf("no recovery within 30s (%d packets blackholed)", blackholed)
|
||||
}
|
||||
return recovered, blackholed
|
||||
}
|
||||
+19
-2
@@ -153,6 +153,9 @@ const (
|
||||
ExitNow ExitType = 1
|
||||
// RouteAndExit routes this packet and exits immediately afterwards
|
||||
RouteAndExit ExitType = 2
|
||||
// Drop discards this packet without delivering it and keeps routing. Use it to simulate a blackhole, such as
|
||||
// a restrictive NAT refusing traffic from an address it has not seen.
|
||||
Drop ExitType = 3
|
||||
)
|
||||
|
||||
type ExitFunc func(packet *udp.Packet, receiver *nebula.Control) ExitType
|
||||
@@ -163,7 +166,9 @@ type ExitFunc func(packet *udp.Packet, receiver *nebula.Control) ExitType
|
||||
func NewR(t testing.TB, controls ...*nebula.Control) *R {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
|
||||
if err := os.MkdirAll("mermaid", 0755); err != nil {
|
||||
// t.Name() contains a slash for subtests, so the flow log can land in a nested directory
|
||||
fn := filepath.Join("mermaid", fmt.Sprintf("%s.md", t.Name()))
|
||||
if err := os.MkdirAll(filepath.Dir(fn), 0755); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
|
||||
@@ -174,7 +179,7 @@ func NewR(t testing.TB, controls ...*nebula.Control) *R {
|
||||
outNat: make(map[outNatKey]netip.AddrPort),
|
||||
flow: []flowEntry{},
|
||||
ignoreFlows: []ignoreFlow{},
|
||||
fn: filepath.Join("mermaid", fmt.Sprintf("%s.md", t.Name())),
|
||||
fn: fn,
|
||||
t: t,
|
||||
cancelRender: cancel,
|
||||
}
|
||||
@@ -687,6 +692,10 @@ func (r *R) RouteExitFunc(sender *nebula.Control, whatDo ExitFunc) {
|
||||
p.Release()
|
||||
return
|
||||
|
||||
case Drop:
|
||||
// Record it so the flow log shows the attempt, but never hand it to the receiver
|
||||
r.unlockedInjectFlow(sender, receiver, p, false)
|
||||
|
||||
case KeepRouting:
|
||||
fp := r.unlockedInjectFlow(sender, receiver, p, false)
|
||||
receiver.InjectUDPPacket(p)
|
||||
@@ -779,6 +788,10 @@ func (r *R) RouteForAllExitFuncOrTimeout(timeout time.Duration, whatDo ExitFunc)
|
||||
p.Release()
|
||||
return true
|
||||
|
||||
case Drop:
|
||||
// Record it so the flow log shows the attempt, but never hand it to the receiver
|
||||
r.unlockedInjectFlow(cm[x], receiver, p, false)
|
||||
|
||||
case KeepRouting:
|
||||
fp := r.unlockedInjectFlow(cm[x], receiver, p, false)
|
||||
receiver.InjectUDPPacket(p)
|
||||
@@ -884,6 +897,10 @@ func (r *R) RouteForAllExitFunc(whatDo ExitFunc) {
|
||||
p.Release()
|
||||
return
|
||||
|
||||
case Drop:
|
||||
// Record it so the flow log shows the attempt, but never hand it to the receiver
|
||||
r.unlockedInjectFlow(cm[x], receiver, p, false)
|
||||
|
||||
case KeepRouting:
|
||||
fp := r.unlockedInjectFlow(cm[x], receiver, p, false)
|
||||
receiver.InjectUDPPacket(p)
|
||||
|
||||
@@ -231,6 +231,35 @@ punchy:
|
||||
# Overriding this to "" is the same as "/" and will allow overwriting any path on the host.
|
||||
#sandbox_dir: /var/tmp/nebula-debug
|
||||
|
||||
# EXPERIMENTAL: this feature may change or disappear in the future.
|
||||
# info_api exposes a small local HTTP+JSON API that lets other programs on
|
||||
# this machine resolve a vpn address to its certificate identity (name, vpn
|
||||
# addresses, groups, fingerprint, validity), e.g. for making authorization
|
||||
# decisions about an inbound connection:
|
||||
# GET /v1/host?addr=<vpn addr> - identity of the host owning the address: a
|
||||
# peer with an active tunnel, or this node itself. `addr` may include a
|
||||
# port (`192.168.100.7:54321`), which is ignored, so a connection's remote
|
||||
# address can be passed through as is. Returns 404 when the address is
|
||||
# unknown or has no active tunnel.
|
||||
# GET /v1/self - this node's own identity.
|
||||
# Identity answers can be trusted because nebula drops inbound packets whose
|
||||
# source vpn address is not contained in the sender's certificate, so the
|
||||
# source address of a connection arriving over the nebula interface is
|
||||
# guaranteed to map to the certificate reported here.
|
||||
# There is no authentication in this API; restrict access with unix socket
|
||||
# file permissions.
|
||||
# This whole section is reloadable.
|
||||
#info_api:
|
||||
# Toggles the feature
|
||||
#enabled: false
|
||||
# listen accepts a unix socket path as a unix:// URL with an absolute path:
|
||||
#listen: unix:///var/run/nebula-info-api.sock
|
||||
# File mode for the unix socket, as an octal string.
|
||||
# The socket is created by nebula's user; to grant a group of local services
|
||||
# access, place the socket in a directory with appropriate permissions
|
||||
# (e.g. a systemd RuntimeDirectory) and relax this to "0660".
|
||||
#socket_mode: "0600"
|
||||
|
||||
# EXPERIMENTAL: relay support for networks that can't establish direct connections.
|
||||
relay:
|
||||
# Relays are a list of Nebula IP's that peers can use to relay packets to me.
|
||||
|
||||
@@ -7,6 +7,7 @@ require (
|
||||
filippo.io/bigmod v0.1.0
|
||||
github.com/anmitsu/go-shlex v0.0.0-20200514113438-38f4b401e2be
|
||||
github.com/armon/go-radix v1.0.0
|
||||
github.com/cyberdelia/go-metrics-graphite v0.0.0-20161219230853-39f87cc3b432
|
||||
github.com/flynn/noise v1.1.0
|
||||
github.com/gaissmai/bart v0.28.0
|
||||
github.com/gogo/protobuf v1.3.2
|
||||
|
||||
@@ -19,6 +19,8 @@ github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6r
|
||||
github.com/cespare/xxhash/v2 v2.1.1/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs=
|
||||
github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs=
|
||||
github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs=
|
||||
github.com/cyberdelia/go-metrics-graphite v0.0.0-20161219230853-39f87cc3b432 h1:M5QgkYacWj0Xs8MhpIK/5uwU02icXpEoSo9sM2aRCps=
|
||||
github.com/cyberdelia/go-metrics-graphite v0.0.0-20161219230853-39f87cc3b432/go.mod h1:xwIwAxMvYnVrGJPe2FKx5prTrnAjGOD8zvDOnxnrrkM=
|
||||
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
|
||||
-117
@@ -1,117 +0,0 @@
|
||||
package nebula
|
||||
|
||||
// This file is a trimmed, inlined copy of the graphite exporter from
|
||||
// github.com/cyberdelia/go-metrics-graphite, retaining only the Config type and
|
||||
// the Once entrypoint that Nebula uses. The upstream package has been
|
||||
// unmaintained for 10+ years, so it was vendored here to drop the dependency.
|
||||
// See https://github.com/slackhq/nebula/issues/1831.
|
||||
//
|
||||
// Copyright 2015 Timothée Peignier. All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are met:
|
||||
//
|
||||
// 1. Redistributions of source code must retain the above copyright notice,
|
||||
// this list of conditions and the following disclaimer.
|
||||
//
|
||||
// 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
// this list of conditions and the following disclaimer in the documentation
|
||||
// and/or other materials provided with the distribution.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
// AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
// IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
// DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
// FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
// DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
// SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
// CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
// OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"fmt"
|
||||
"net"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/rcrowley/go-metrics"
|
||||
)
|
||||
|
||||
// graphiteConfigExport provides a container with configuration parameters for
|
||||
// the Graphite exporter.
|
||||
type graphiteConfigExport struct {
|
||||
Addr *net.TCPAddr // Network address to connect to
|
||||
Registry metrics.Registry // Registry to be exported
|
||||
FlushInterval time.Duration // Flush interval
|
||||
DurationUnit time.Duration // Time conversion unit for durations
|
||||
Prefix string // Prefix to be prepended to metric names
|
||||
Percentiles []float64 // Percentiles to export from timers and histograms
|
||||
}
|
||||
|
||||
// graphiteOnce performs a single submission to Graphite, returning a non-nil
|
||||
// error on failed connections.
|
||||
func graphiteOnce(c graphiteConfigExport) error {
|
||||
now := time.Now().Unix()
|
||||
du := float64(c.DurationUnit)
|
||||
flushSeconds := float64(c.FlushInterval) / float64(time.Second)
|
||||
conn, err := net.DialTCP("tcp", nil, c.Addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer conn.Close()
|
||||
w := bufio.NewWriter(conn)
|
||||
c.Registry.Each(func(name string, i any) {
|
||||
switch metric := i.(type) {
|
||||
case metrics.Counter:
|
||||
count := metric.Count()
|
||||
fmt.Fprintf(w, "%s.%s.count %d %d\n", c.Prefix, name, count, now)
|
||||
fmt.Fprintf(w, "%s.%s.count_ps %.2f %d\n", c.Prefix, name, float64(count)/flushSeconds, now)
|
||||
case metrics.Gauge:
|
||||
fmt.Fprintf(w, "%s.%s.value %d %d\n", c.Prefix, name, metric.Value(), now)
|
||||
case metrics.GaugeFloat64:
|
||||
fmt.Fprintf(w, "%s.%s.value %f %d\n", c.Prefix, name, metric.Value(), now)
|
||||
case metrics.Histogram:
|
||||
h := metric.Snapshot()
|
||||
ps := h.Percentiles(c.Percentiles)
|
||||
fmt.Fprintf(w, "%s.%s.count %d %d\n", c.Prefix, name, h.Count(), now)
|
||||
fmt.Fprintf(w, "%s.%s.min %d %d\n", c.Prefix, name, h.Min(), now)
|
||||
fmt.Fprintf(w, "%s.%s.max %d %d\n", c.Prefix, name, h.Max(), now)
|
||||
fmt.Fprintf(w, "%s.%s.mean %.2f %d\n", c.Prefix, name, h.Mean(), now)
|
||||
fmt.Fprintf(w, "%s.%s.std-dev %.2f %d\n", c.Prefix, name, h.StdDev(), now)
|
||||
for psIdx, psKey := range c.Percentiles {
|
||||
key := strings.Replace(strconv.FormatFloat(psKey*100.0, 'f', -1, 64), ".", "", 1)
|
||||
fmt.Fprintf(w, "%s.%s.%s-percentile %.2f %d\n", c.Prefix, name, key, ps[psIdx], now)
|
||||
}
|
||||
case metrics.Meter:
|
||||
m := metric.Snapshot()
|
||||
fmt.Fprintf(w, "%s.%s.count %d %d\n", c.Prefix, name, m.Count(), now)
|
||||
fmt.Fprintf(w, "%s.%s.one-minute %.2f %d\n", c.Prefix, name, m.Rate1(), now)
|
||||
fmt.Fprintf(w, "%s.%s.five-minute %.2f %d\n", c.Prefix, name, m.Rate5(), now)
|
||||
fmt.Fprintf(w, "%s.%s.fifteen-minute %.2f %d\n", c.Prefix, name, m.Rate15(), now)
|
||||
fmt.Fprintf(w, "%s.%s.mean %.2f %d\n", c.Prefix, name, m.RateMean(), now)
|
||||
case metrics.Timer:
|
||||
t := metric.Snapshot()
|
||||
ps := t.Percentiles(c.Percentiles)
|
||||
count := t.Count()
|
||||
fmt.Fprintf(w, "%s.%s.count %d %d\n", c.Prefix, name, count, now)
|
||||
fmt.Fprintf(w, "%s.%s.count_ps %.2f %d\n", c.Prefix, name, float64(count)/flushSeconds, now)
|
||||
fmt.Fprintf(w, "%s.%s.min %d %d\n", c.Prefix, name, t.Min()/int64(du), now)
|
||||
fmt.Fprintf(w, "%s.%s.max %d %d\n", c.Prefix, name, t.Max()/int64(du), now)
|
||||
fmt.Fprintf(w, "%s.%s.mean %.2f %d\n", c.Prefix, name, t.Mean()/du, now)
|
||||
fmt.Fprintf(w, "%s.%s.std-dev %.2f %d\n", c.Prefix, name, t.StdDev()/du, now)
|
||||
for psIdx, psKey := range c.Percentiles {
|
||||
key := strings.Replace(strconv.FormatFloat(psKey*100.0, 'f', -1, 64), ".", "", 1)
|
||||
fmt.Fprintf(w, "%s.%s.%s-percentile %.2f %d\n", c.Prefix, name, key, ps[psIdx]/du, now)
|
||||
}
|
||||
fmt.Fprintf(w, "%s.%s.one-minute %.2f %d\n", c.Prefix, name, t.Rate1(), now)
|
||||
fmt.Fprintf(w, "%s.%s.five-minute %.2f %d\n", c.Prefix, name, t.Rate5(), now)
|
||||
fmt.Fprintf(w, "%s.%s.fifteen-minute %.2f %d\n", c.Prefix, name, t.Rate15(), now)
|
||||
fmt.Fprintf(w, "%s.%s.mean-rate %.2f %d\n", c.Prefix, name, t.RateMean(), now)
|
||||
}
|
||||
w.Flush()
|
||||
})
|
||||
return nil
|
||||
}
|
||||
+2
-14
@@ -749,14 +749,8 @@ func (hm *HandshakeManager) beginHandshake(via ViaSender, packet []byte, h *head
|
||||
return
|
||||
}
|
||||
|
||||
connState, err := newConnectionStateFromResult(result)
|
||||
if err != nil {
|
||||
f.l.Error("Discarding handshake with an invalid message index", "error", err, "vpnAddrs", vpnAddrs)
|
||||
return
|
||||
}
|
||||
|
||||
hostinfo := &HostInfo{
|
||||
ConnectionState: connState,
|
||||
ConnectionState: newConnectionStateFromResult(result),
|
||||
localIndexId: result.LocalIndex,
|
||||
remoteIndexId: result.RemoteIndex,
|
||||
vpnAddrs: vpnAddrs,
|
||||
@@ -874,13 +868,7 @@ func (hm *HandshakeManager) continueHandshake(via ViaSender, hh *HandshakeHostIn
|
||||
}
|
||||
|
||||
// Handshake complete; build the ConnectionState now that we have keys and a verified peer cert.
|
||||
cs, err := newConnectionStateFromResult(result)
|
||||
if err != nil {
|
||||
f.l.Error("Discarding handshake with an invalid message index", "error", err, "vpnAddrs", hostinfo.vpnAddrs)
|
||||
hm.DeleteHostInfo(hostinfo)
|
||||
return
|
||||
}
|
||||
hostinfo.ConnectionState = cs
|
||||
hostinfo.ConnectionState = newConnectionStateFromResult(result)
|
||||
|
||||
remoteCert := result.RemoteCert
|
||||
if remoteCert == nil {
|
||||
|
||||
@@ -0,0 +1,409 @@
|
||||
package nebula
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"log/slog"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/netip"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/slackhq/nebula/cert"
|
||||
"github.com/slackhq/nebula/config"
|
||||
)
|
||||
|
||||
// infoAPIServer is a small http+json listener on a unix socket that lets other
|
||||
// programs on this machine resolve a vpn address to its certificate identity (name, groups, networks)
|
||||
// for making authorization decisions. Lifecycle works like statsServer: the constructor wires the
|
||||
// reload callback, reload records config, Start runs the runtime, Stop tears it down
|
||||
type infoAPIServer struct {
|
||||
l *slog.Logger
|
||||
ctx context.Context
|
||||
hostMap *HostMap
|
||||
pki *PKI
|
||||
|
||||
// enabled mirrors `info_api.enabled` so callers of Start don't need to know the gating rules
|
||||
enabled atomic.Bool
|
||||
|
||||
runMu sync.Mutex
|
||||
runCfg *infoAPIConfig
|
||||
run *infoAPIRuntime // non-nil while a runtime is live
|
||||
}
|
||||
|
||||
// infoAPIRuntime is the live state owned by a single Start invocation. Stop and Start's exit path
|
||||
// use pointer equality to tell "my runtime" apart from one that replaced it after a reload
|
||||
type infoAPIRuntime struct {
|
||||
server *http.Server
|
||||
listener net.Listener
|
||||
}
|
||||
|
||||
// infoAPIConfig is a snapshot of the info_api config section, comparable with == so reload can
|
||||
// detect "no change" cheaply
|
||||
type infoAPIConfig struct {
|
||||
enabled bool
|
||||
listen string // raw config value, for error messages
|
||||
addr string // unix socket path
|
||||
// file mode applied to the unix socket after bind
|
||||
socketMode fs.FileMode
|
||||
}
|
||||
|
||||
// newInfoAPIServerFromConfig builds a infoAPIServer and applies the initial config. The reload
|
||||
// callback is registered first so a SIGHUP can later enable, fix, or disable the listener even if
|
||||
// the initial config was bad. Nothing binds until Start, so config tests are side effect free.
|
||||
// A bad config is logged rather than returned: it must not stop nebula from starting, the feature
|
||||
// just stays disabled until a reload provides a valid config
|
||||
func newInfoAPIServerFromConfig(ctx context.Context, l *slog.Logger, pki *PKI, hostMap *HostMap, c *config.C) *infoAPIServer {
|
||||
h := &infoAPIServer{
|
||||
l: l,
|
||||
ctx: ctx,
|
||||
hostMap: hostMap,
|
||||
pki: pki,
|
||||
}
|
||||
|
||||
c.RegisterReloadCallback(func(c *config.C) {
|
||||
if err := h.reload(c, false); err != nil {
|
||||
h.l.Warn("Failed to reload info API from config", "error", err)
|
||||
}
|
||||
})
|
||||
|
||||
if err := h.reload(c, true); err != nil {
|
||||
h.l.Warn("Failed to apply info API config; it will stay disabled until the config is fixed and reloaded", "error", err)
|
||||
}
|
||||
return h
|
||||
}
|
||||
|
||||
// reload records the latest config. The initial call only records it, Control.Start launches the
|
||||
// first runtime via infoAPIStart. Later calls reconcile the running listener with the new config:
|
||||
// enable, disable, or restart when the listen config changed
|
||||
func (h *infoAPIServer) reload(c *config.C, initial bool) error {
|
||||
newCfg, err := loadInfoAPIConfig(c)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
h.runMu.Lock()
|
||||
sameCfg := h.runCfg != nil && *h.runCfg == newCfg
|
||||
h.runCfg = &newCfg
|
||||
running := h.run != nil
|
||||
h.runMu.Unlock()
|
||||
|
||||
h.enabled.Store(newCfg.enabled)
|
||||
|
||||
if initial || sameCfg {
|
||||
return nil
|
||||
}
|
||||
|
||||
if running {
|
||||
h.Stop()
|
||||
}
|
||||
if newCfg.enabled {
|
||||
go h.Start()
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Start binds the listener from the latest config and serves until Stop is called or ctx fires.
|
||||
// Safe to call when disabled or already running (both no-op)
|
||||
func (h *infoAPIServer) Start() {
|
||||
if !h.enabled.Load() {
|
||||
return
|
||||
}
|
||||
|
||||
h.runMu.Lock()
|
||||
if h.ctx.Err() != nil || h.run != nil || h.runCfg == nil {
|
||||
h.runMu.Unlock()
|
||||
return
|
||||
}
|
||||
cfg := *h.runCfg
|
||||
ln, err := h.listen(cfg)
|
||||
if err != nil {
|
||||
// drop the cached config so a SIGHUP with the same config retries the bind
|
||||
h.runCfg = nil
|
||||
h.runMu.Unlock()
|
||||
h.l.Error("Failed to start info API listener", "listen", cfg.listen, "error", err)
|
||||
return
|
||||
}
|
||||
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("GET /v1/host", h.handleHost)
|
||||
mux.HandleFunc("GET /v1/self", h.handleSelf)
|
||||
srv := &http.Server{Handler: mux, ReadHeaderTimeout: 5 * time.Second}
|
||||
rt := &infoAPIRuntime{server: srv, listener: ln}
|
||||
h.run = rt
|
||||
h.runMu.Unlock()
|
||||
|
||||
h.l.Info("Starting info API listener", "addr", ln.Addr())
|
||||
cleanExit := h.serve(srv, ln)
|
||||
|
||||
// A Stop that raced our bind shut the server down before Serve could adopt the listener;
|
||||
// closing it again is harmless and guarantees a unix socket file gets unlinked
|
||||
_ = ln.Close()
|
||||
|
||||
// Clear our runtime only if nothing has replaced it. Stop races through here too but leaves
|
||||
// h.run == nil, so the pointer check skips
|
||||
h.runMu.Lock()
|
||||
if h.run == rt {
|
||||
h.run = nil
|
||||
// an error exit leaves runCfg cached as if it were applied, drop it so a SIGHUP with the
|
||||
// same config re-triggers Start once the user fixes the underlying problem
|
||||
if !cleanExit {
|
||||
h.runCfg = nil
|
||||
}
|
||||
}
|
||||
h.runMu.Unlock()
|
||||
}
|
||||
|
||||
// serve runs srv.Serve and ensures ctx cancellation unblocks it. Returns true if the listener
|
||||
// exited cleanly (Stop, ctx cancellation), false on an unexpected error
|
||||
func (h *infoAPIServer) serve(srv *http.Server, ln net.Listener) bool {
|
||||
// ctx cancellation triggers a server shutdown which in turn unblocks Serve, closing `done` on
|
||||
// exit keeps the watcher from outliving this call
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
select {
|
||||
case <-h.ctx.Done():
|
||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
if err := srv.Shutdown(shutdownCtx); err != nil {
|
||||
h.l.Warn("Failed to shut down info API listener", "error", err)
|
||||
}
|
||||
case <-done:
|
||||
}
|
||||
}()
|
||||
defer close(done)
|
||||
|
||||
err := srv.Serve(ln)
|
||||
if err == nil || errors.Is(err, http.ErrServerClosed) {
|
||||
return true
|
||||
}
|
||||
h.l.Error("Info API listener exited", "error", err)
|
||||
return false
|
||||
}
|
||||
|
||||
// Stop tears down the active runtime, if any. Idempotent
|
||||
func (h *infoAPIServer) Stop() {
|
||||
h.runMu.Lock()
|
||||
rt := h.run
|
||||
h.run = nil
|
||||
h.runMu.Unlock()
|
||||
if rt == nil {
|
||||
return
|
||||
}
|
||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
if err := rt.server.Shutdown(shutdownCtx); err != nil {
|
||||
h.l.Warn("Failed to shut down info API listener", "error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// listen binds the configured unix socket. It also clears a stale socket file left by an unclean
|
||||
// exit and applies the configured file mode
|
||||
func (h *infoAPIServer) listen(cfg infoAPIConfig) (net.Listener, error) {
|
||||
if fi, err := os.Stat(cfg.addr); err == nil {
|
||||
if fi.Mode()&os.ModeSocket == 0 {
|
||||
return nil, fmt.Errorf("info_api.listen path %s exists and is not a socket, refusing to replace it", cfg.addr)
|
||||
}
|
||||
// a normal shutdown unlinks the socket, so a file here means a previous process exited
|
||||
// uncleanly, remove it so the bind below can succeed
|
||||
if err = os.Remove(cfg.addr); err != nil {
|
||||
return nil, fmt.Errorf("failed to remove stale socket %s: %w", cfg.addr, err)
|
||||
}
|
||||
}
|
||||
|
||||
ln, err := net.Listen("unix", cfg.addr)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// The socket is briefly live with umask-derived permissions before this chmod lands, tolerated
|
||||
// because connections accepted in that window still only reach this read-only API
|
||||
if err = os.Chmod(cfg.addr, cfg.socketMode); err != nil {
|
||||
_ = ln.Close()
|
||||
return nil, fmt.Errorf("failed to set mode on socket %s: %w", cfg.addr, err)
|
||||
}
|
||||
return ln, nil
|
||||
}
|
||||
|
||||
func (h *infoAPIServer) certState() *CertState {
|
||||
if h.pki == nil {
|
||||
return nil
|
||||
}
|
||||
return h.pki.getCertState()
|
||||
}
|
||||
|
||||
// handleHost serves GET /v1/host?addr=<vpn addr>, answering with the identity of the host that
|
||||
// owns the address: a peer with an active tunnel, or this node itself. addr may include a port,
|
||||
// which is ignored, so clients can pass a connection's remote address through without parsing it
|
||||
func (h *infoAPIServer) handleHost(w http.ResponseWriter, r *http.Request) {
|
||||
q := r.URL.Query().Get("addr")
|
||||
if q == "" {
|
||||
writeJSONError(w, http.StatusBadRequest, "missing addr parameter")
|
||||
return
|
||||
}
|
||||
ip, err := parseQueryAddrParam(q)
|
||||
if err != nil {
|
||||
writeJSONError(w, http.StatusBadRequest, "invalid address")
|
||||
return
|
||||
}
|
||||
|
||||
crt := findCertificateForVpnAddr(h.certState(), h.hostMap, ip)
|
||||
if crt == nil {
|
||||
writeJSONError(w, http.StatusNotFound, "no active tunnel for address")
|
||||
return
|
||||
}
|
||||
h.writeHostIdentity(w, crt)
|
||||
}
|
||||
|
||||
// handleSelf serves GET /v1/self, answering with this node's own identity
|
||||
func (h *infoAPIServer) handleSelf(w http.ResponseWriter, r *http.Request) {
|
||||
var crt cert.Certificate
|
||||
if cs := h.certState(); cs != nil {
|
||||
crt = cs.getCertificate(cs.initiatingVersion)
|
||||
}
|
||||
if crt == nil {
|
||||
writeJSONError(w, http.StatusInternalServerError, "no certificate available")
|
||||
return
|
||||
}
|
||||
h.writeHostIdentity(w, crt)
|
||||
}
|
||||
|
||||
func (h *infoAPIServer) writeHostIdentity(w http.ResponseWriter, crt cert.Certificate) {
|
||||
id, err := newHostIdentity(crt)
|
||||
if err != nil {
|
||||
writeJSONError(w, http.StatusInternalServerError, "failed to fingerprint certificate")
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
if err = json.NewEncoder(w).Encode(id); err != nil {
|
||||
h.l.Debug("Failed to write info API response", "error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// findCertificateForVpnAddr answers "who owns this vpn address": ourselves (from local cert state,
|
||||
// the hostmap never carries an entry for this node) or a peer with an active tunnel. Returns nil
|
||||
// when the address is unknown or the tunnel is mid-teardown
|
||||
func findCertificateForVpnAddr(cs *CertState, hostMap *HostMap, ip netip.Addr) cert.Certificate {
|
||||
if cs != nil && cs.myVpnAddrsTable != nil && cs.myVpnAddrsTable.Contains(ip) {
|
||||
return cs.getCertificate(cs.initiatingVersion)
|
||||
}
|
||||
|
||||
hostinfo := hostMap.QueryVpnAddr(ip)
|
||||
if hostinfo == nil {
|
||||
return nil
|
||||
}
|
||||
cc := hostinfo.GetCert()
|
||||
if cc == nil {
|
||||
return nil
|
||||
}
|
||||
return cc.Certificate
|
||||
}
|
||||
|
||||
// hostIdentity is the json document served for both /v1/host and /v1/self, every field is derived
|
||||
// from the authenticated certificate alone
|
||||
type hostIdentity struct {
|
||||
Name string `json:"name"`
|
||||
VpnAddrs []netip.Addr `json:"vpnAddrs"`
|
||||
Networks []netip.Prefix `json:"networks"`
|
||||
UnsafeNetworks []netip.Prefix `json:"unsafeNetworks"`
|
||||
Groups []string `json:"groups"`
|
||||
Fingerprint string `json:"fingerprint"`
|
||||
Issuer string `json:"issuer"`
|
||||
NotBefore time.Time `json:"notBefore"`
|
||||
NotAfter time.Time `json:"notAfter"`
|
||||
CertVersion int `json:"certVersion"`
|
||||
}
|
||||
|
||||
func newHostIdentity(crt cert.Certificate) (hostIdentity, error) {
|
||||
fp, err := crt.Fingerprint()
|
||||
if err != nil {
|
||||
return hostIdentity{}, err
|
||||
}
|
||||
|
||||
// slices are always allocated so they marshal as [] rather than null
|
||||
networks := crt.Networks()
|
||||
id := hostIdentity{
|
||||
Name: crt.Name(),
|
||||
VpnAddrs: make([]netip.Addr, 0, len(networks)),
|
||||
Networks: append(make([]netip.Prefix, 0, len(networks)), networks...),
|
||||
UnsafeNetworks: append(make([]netip.Prefix, 0, len(crt.UnsafeNetworks())), crt.UnsafeNetworks()...),
|
||||
Groups: append(make([]string, 0, len(crt.Groups())), crt.Groups()...),
|
||||
Fingerprint: fp,
|
||||
Issuer: crt.Issuer(),
|
||||
NotBefore: crt.NotBefore(),
|
||||
NotAfter: crt.NotAfter(),
|
||||
CertVersion: int(crt.Version()),
|
||||
}
|
||||
for _, n := range networks {
|
||||
id.VpnAddrs = append(id.VpnAddrs, n.Addr())
|
||||
}
|
||||
return id, nil
|
||||
}
|
||||
|
||||
func writeJSONError(w http.ResponseWriter, status int, msg string) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.WriteHeader(status)
|
||||
_ = json.NewEncoder(w).Encode(map[string]string{"error": msg})
|
||||
}
|
||||
|
||||
// parseQueryAddrParam parses the addr query parameter, accepting a bare address or an address with
|
||||
// a port (`192.168.100.7:54321`, `[fd00::1]:443`) so callers can pass a connection's RemoteAddr
|
||||
// straight through. The result is unmapped, 4in6 addresses (::ffff:a.b.c.d) become ipv4
|
||||
func parseQueryAddrParam(s string) (netip.Addr, error) {
|
||||
if ip, err := netip.ParseAddr(s); err == nil {
|
||||
return ip.Unmap(), nil
|
||||
}
|
||||
ap, err := netip.ParseAddrPort(s)
|
||||
if err != nil {
|
||||
return netip.Addr{}, err
|
||||
}
|
||||
return ap.Addr().Unmap(), nil
|
||||
}
|
||||
|
||||
func loadInfoAPIConfig(c *config.C) (infoAPIConfig, error) {
|
||||
cfg := infoAPIConfig{
|
||||
enabled: c.GetBool("info_api.enabled", false),
|
||||
listen: c.GetString("info_api.listen", ""),
|
||||
}
|
||||
if !cfg.enabled {
|
||||
return cfg, nil
|
||||
}
|
||||
|
||||
if cfg.listen == "" {
|
||||
return cfg, errors.New("info_api.listen can not be empty when info_api is enabled")
|
||||
}
|
||||
addr, err := parseInfoAPIListen(cfg.listen)
|
||||
if err != nil {
|
||||
return cfg, err
|
||||
}
|
||||
cfg.addr = addr
|
||||
|
||||
// read as a string so yaml can't reinterpret the octal literal
|
||||
modeStr := c.GetString("info_api.socket_mode", "0600")
|
||||
mode, err := strconv.ParseUint(modeStr, 8, 32)
|
||||
if err != nil || fs.FileMode(mode)&^fs.ModePerm != 0 {
|
||||
return cfg, fmt.Errorf("info_api.socket_mode was not a valid octal file mode: %s", modeStr)
|
||||
}
|
||||
cfg.socketMode = fs.FileMode(mode)
|
||||
return cfg, nil
|
||||
}
|
||||
|
||||
// parseInfoAPIListen extracts the unix socket path from the info_api.listen config value, which
|
||||
// must be a `unix://` URL with an absolute path, e.g. `unix:///var/run/nebula.sock`
|
||||
func parseInfoAPIListen(listen string) (addr string, err error) {
|
||||
path, ok := strings.CutPrefix(listen, "unix://")
|
||||
if !ok {
|
||||
return "", fmt.Errorf("info_api.listen must be a unix:// socket path: %s", listen)
|
||||
} else if !filepath.IsAbs(path) {
|
||||
return "", fmt.Errorf("info_api.listen unix socket path must be absolute: %s", listen)
|
||||
}
|
||||
return path, nil
|
||||
}
|
||||
@@ -0,0 +1,448 @@
|
||||
package nebula
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"log/slog"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/netip"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/slackhq/nebula/cert"
|
||||
"github.com/slackhq/nebula/cert_test"
|
||||
"github.com/slackhq/nebula/config"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func Test_parseInfoAPIListen(t *testing.T) {
|
||||
type testCase struct {
|
||||
listen string
|
||||
addr string
|
||||
wantErr bool
|
||||
}
|
||||
tests := []testCase{
|
||||
{listen: "", wantErr: true},
|
||||
{listen: "unix://", wantErr: true},
|
||||
{listen: "unix://relative/path.sock", wantErr: true},
|
||||
{listen: "not an address", wantErr: true},
|
||||
// tcp host:port addresses are no longer accepted
|
||||
{listen: "127.0.0.1:8085", wantErr: true},
|
||||
{listen: "[::1]:8085", wantErr: true},
|
||||
{listen: "localhost:8085", wantErr: true},
|
||||
}
|
||||
|
||||
// A unix socket path must be absolute for the OS that will bind it, and filepath.IsAbs is
|
||||
// GOOS-specific. CI runs the suite separately on each OS, so assert the platform's own native
|
||||
// absolute path is accepted while the other platform's is rejected.
|
||||
posixPath := "unix:///var/run/nebula.sock"
|
||||
winPath := `unix://C:\nebula\hq.sock`
|
||||
if runtime.GOOS == "windows" {
|
||||
tests = append(tests,
|
||||
testCase{listen: winPath, addr: `C:\nebula\hq.sock`},
|
||||
testCase{listen: posixPath, wantErr: true},
|
||||
)
|
||||
} else {
|
||||
tests = append(tests,
|
||||
testCase{listen: posixPath, addr: "/var/run/nebula.sock"},
|
||||
testCase{listen: winPath, wantErr: true},
|
||||
)
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
addr, err := parseInfoAPIListen(tt.listen)
|
||||
if tt.wantErr {
|
||||
require.Error(t, err, "listen=%q", tt.listen)
|
||||
continue
|
||||
}
|
||||
require.NoError(t, err, "listen=%q", tt.listen)
|
||||
assert.Equal(t, tt.addr, addr, "listen=%q", tt.listen)
|
||||
}
|
||||
}
|
||||
|
||||
func Test_loadInfoAPIConfig(t *testing.T) {
|
||||
c := config.NewC(nil)
|
||||
|
||||
// the listen path must be absolute for the OS running the test (CI is per-OS)
|
||||
listen, wantAddr := "unix:///tmp/hq.sock", "/tmp/hq.sock"
|
||||
if runtime.GOOS == "windows" {
|
||||
listen, wantAddr = `unix://C:\tmp\hq.sock`, `C:\tmp\hq.sock`
|
||||
}
|
||||
|
||||
// absent section means disabled, no error
|
||||
cfg, err := loadInfoAPIConfig(c)
|
||||
require.NoError(t, err)
|
||||
assert.False(t, cfg.enabled)
|
||||
|
||||
// enabled without a listen address is an error
|
||||
setInfoAPIConfig(c, true, "", "")
|
||||
_, err = loadInfoAPIConfig(c)
|
||||
require.Error(t, err)
|
||||
|
||||
// a unix socket gets the default mode
|
||||
setInfoAPIConfig(c, true, listen, "")
|
||||
cfg, err = loadInfoAPIConfig(c)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, wantAddr, cfg.addr)
|
||||
assert.Equal(t, fs.FileMode(0o600), cfg.socketMode)
|
||||
|
||||
setInfoAPIConfig(c, true, listen, "0660")
|
||||
cfg, err = loadInfoAPIConfig(c)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, fs.FileMode(0o660), cfg.socketMode)
|
||||
|
||||
setInfoAPIConfig(c, true, listen, "withers")
|
||||
_, err = loadInfoAPIConfig(c)
|
||||
require.Error(t, err)
|
||||
|
||||
// mode bits beyond the permission bits are rejected
|
||||
setInfoAPIConfig(c, true, listen, "10600")
|
||||
_, err = loadInfoAPIConfig(c)
|
||||
require.Error(t, err)
|
||||
|
||||
// tcp host:port listen addresses are no longer supported
|
||||
setInfoAPIConfig(c, true, "127.0.0.1:8085", "")
|
||||
_, err = loadInfoAPIConfig(c)
|
||||
require.Error(t, err)
|
||||
}
|
||||
|
||||
func TestInfoAPIServer_badConfigIsNonFatal(t *testing.T) {
|
||||
// an enabled-but-invalid config must not stop construction; nebula keeps starting and the
|
||||
// feature simply stays disabled until a reload supplies a valid config
|
||||
c := config.NewC(nil)
|
||||
setInfoAPIConfig(c, true, "not-a-unix-socket", "")
|
||||
h := newInfoAPIServerFromConfig(context.Background(), slog.New(slog.DiscardHandler), nil, newHostMap(slog.New(slog.DiscardHandler)), c)
|
||||
require.NotNil(t, h)
|
||||
assert.False(t, h.enabled.Load())
|
||||
|
||||
// no config was recorded, so Start has nothing to bind and is a no-op
|
||||
h.runMu.Lock()
|
||||
assert.Nil(t, h.runCfg)
|
||||
h.runMu.Unlock()
|
||||
h.Start()
|
||||
h.runMu.Lock()
|
||||
assert.Nil(t, h.run)
|
||||
h.runMu.Unlock()
|
||||
}
|
||||
|
||||
func setInfoAPIConfig(c *config.C, enabled bool, listen, socketMode string) {
|
||||
settings := map[string]any{
|
||||
"enabled": enabled,
|
||||
"listen": listen,
|
||||
}
|
||||
if socketMode != "" {
|
||||
settings["socket_mode"] = socketMode
|
||||
}
|
||||
c.Settings["info_api"] = settings
|
||||
}
|
||||
|
||||
func newTestInfoAPIServer(t *testing.T) (*infoAPIServer, *config.C) {
|
||||
t.Helper()
|
||||
h := &infoAPIServer{
|
||||
l: slog.New(slog.DiscardHandler),
|
||||
ctx: context.Background(),
|
||||
hostMap: newHostMap(slog.New(slog.DiscardHandler)),
|
||||
}
|
||||
h.hostMap.preferredRanges.Store(&[]netip.Prefix{})
|
||||
return h, config.NewC(nil)
|
||||
}
|
||||
|
||||
// addTestPeer creates a certificate for a peer owning each addr (as a /24 or /64) and inserts it
|
||||
// into the hostmap as an established tunnel
|
||||
func addTestPeer(t *testing.T, hm *HostMap, name string, addrs []netip.Addr, unsafeNetworks []netip.Prefix, groups []string) cert.Certificate {
|
||||
t.Helper()
|
||||
networks := make([]netip.Prefix, 0, len(addrs))
|
||||
for _, a := range addrs {
|
||||
bits := 24
|
||||
if a.Is6() {
|
||||
bits = 64
|
||||
}
|
||||
networks = append(networks, netip.PrefixFrom(a, bits))
|
||||
}
|
||||
ca, _, caKey, _ := cert_test.NewTestCaCert(cert.Version2, cert.Curve_CURVE25519, time.Time{}, time.Time{}, nil, nil, nil)
|
||||
crt, _, _, _ := cert_test.NewTestCert(cert.Version2, cert.Curve_CURVE25519, ca, caKey, name, time.Time{}, time.Time{}, networks, unsafeNetworks, groups)
|
||||
fp, err := crt.Fingerprint()
|
||||
require.NoError(t, err)
|
||||
|
||||
hm.unlockedAddHostInfo(&HostInfo{
|
||||
ConnectionState: &ConnectionState{
|
||||
peerCert: &cert.CachedCertificate{Certificate: crt, Fingerprint: fp},
|
||||
},
|
||||
vpnAddrs: addrs,
|
||||
relayState: RelayState{
|
||||
relayForByAddr: map[netip.Addr]*Relay{},
|
||||
relayForByIdx: map[uint32]*Relay{},
|
||||
},
|
||||
}, &Interface{})
|
||||
return crt
|
||||
}
|
||||
|
||||
func getHost(t *testing.T, h *infoAPIServer, addrParam string) (int, map[string]any) {
|
||||
t.Helper()
|
||||
r := httptest.NewRequest(http.MethodGet, "/v1/host?addr="+url.QueryEscape(addrParam), nil)
|
||||
w := httptest.NewRecorder()
|
||||
h.handleHost(w, r)
|
||||
return decodeResponse(t, w)
|
||||
}
|
||||
|
||||
func decodeResponse(t *testing.T, w *httptest.ResponseRecorder) (int, map[string]any) {
|
||||
t.Helper()
|
||||
assert.Equal(t, "application/json", w.Header().Get("Content-Type"))
|
||||
var body map[string]any
|
||||
require.NoError(t, json.Unmarshal(w.Body.Bytes(), &body))
|
||||
return w.Code, body
|
||||
}
|
||||
|
||||
func TestInfoAPIServer_handleHost(t *testing.T) {
|
||||
h, _ := newTestInfoAPIServer(t)
|
||||
h.pki = newTestPKI(t, "self", []netip.Addr{netip.MustParseAddr("10.0.0.1")})
|
||||
|
||||
peerV4 := netip.MustParseAddr("10.0.0.99")
|
||||
peerV6 := netip.MustParseAddr("fd00::99")
|
||||
addTestPeer(t, h.hostMap, "laptop-alice", []netip.Addr{peerV4, peerV6},
|
||||
[]netip.Prefix{netip.MustParsePrefix("192.168.50.0/24")}, []string{"eng", "ssh"})
|
||||
addTestPeer(t, h.hostMap, "groupless", []netip.Addr{netip.MustParseAddr("10.0.0.77")}, nil, nil)
|
||||
|
||||
// an established peer comes back with its full identity
|
||||
code, body := getHost(t, h, "10.0.0.99")
|
||||
require.Equal(t, http.StatusOK, code)
|
||||
assert.Equal(t, "laptop-alice", body["name"])
|
||||
assert.Equal(t, []any{"10.0.0.99", "fd00::99"}, body["vpnAddrs"])
|
||||
assert.Equal(t, []any{"10.0.0.99/24", "fd00::99/64"}, body["networks"])
|
||||
assert.Equal(t, []any{"192.168.50.0/24"}, body["unsafeNetworks"])
|
||||
assert.Equal(t, []any{"eng", "ssh"}, body["groups"])
|
||||
assert.NotEmpty(t, body["fingerprint"])
|
||||
assert.Equal(t, "2", fmt.Sprintf("%v", body["certVersion"]))
|
||||
assert.NotEmpty(t, body["notBefore"])
|
||||
assert.NotEmpty(t, body["notAfter"])
|
||||
|
||||
// empty cert slices marshal as [] rather than null
|
||||
code, body = getHost(t, h, "10.0.0.77")
|
||||
require.Equal(t, http.StatusOK, code)
|
||||
require.NotNil(t, body["groups"])
|
||||
assert.Empty(t, body["groups"])
|
||||
require.NotNil(t, body["unsafeNetworks"])
|
||||
assert.Empty(t, body["unsafeNetworks"])
|
||||
|
||||
// a port in addr is ignored so RemoteAddr can be passed through directly, including the
|
||||
// bracketed v6 and 4in6 forms
|
||||
for _, q := range []string{"10.0.0.99:54321", "[fd00::99]:443", "::ffff:10.0.0.99"} {
|
||||
code, body = getHost(t, h, q)
|
||||
require.Equal(t, http.StatusOK, code, "addr=%q", q)
|
||||
assert.Equal(t, "laptop-alice", body["name"], "addr=%q", q)
|
||||
}
|
||||
|
||||
// our own address answers from the local cert state
|
||||
code, body = getHost(t, h, "10.0.0.1")
|
||||
require.Equal(t, http.StatusOK, code)
|
||||
assert.Equal(t, "self", body["name"])
|
||||
|
||||
code, body = getHost(t, h, "10.0.0.42")
|
||||
assert.Equal(t, http.StatusNotFound, code)
|
||||
assert.NotEmpty(t, body["error"])
|
||||
|
||||
// a tunnel mid-teardown (no peer cert) is treated as unknown
|
||||
h.hostMap.unlockedAddHostInfo(&HostInfo{
|
||||
ConnectionState: &ConnectionState{},
|
||||
vpnAddrs: []netip.Addr{netip.MustParseAddr("10.0.0.66")},
|
||||
relayState: RelayState{
|
||||
relayForByAddr: map[netip.Addr]*Relay{},
|
||||
relayForByIdx: map[uint32]*Relay{},
|
||||
},
|
||||
}, &Interface{})
|
||||
code, _ = getHost(t, h, "10.0.0.66")
|
||||
assert.Equal(t, http.StatusNotFound, code)
|
||||
|
||||
code, body = getHost(t, h, "not-an-address")
|
||||
assert.Equal(t, http.StatusBadRequest, code)
|
||||
assert.NotEmpty(t, body["error"])
|
||||
|
||||
r := httptest.NewRequest(http.MethodGet, "/v1/host", nil)
|
||||
w := httptest.NewRecorder()
|
||||
h.handleHost(w, r)
|
||||
code, body = decodeResponse(t, w)
|
||||
assert.Equal(t, http.StatusBadRequest, code)
|
||||
assert.NotEmpty(t, body["error"])
|
||||
}
|
||||
|
||||
func TestInfoAPIServer_handleSelf(t *testing.T) {
|
||||
h, _ := newTestInfoAPIServer(t)
|
||||
h.pki = newTestPKI(t, "lighthouse", []netip.Addr{netip.MustParseAddr("10.0.0.1")})
|
||||
|
||||
r := httptest.NewRequest(http.MethodGet, "/v1/self", nil)
|
||||
w := httptest.NewRecorder()
|
||||
h.handleSelf(w, r)
|
||||
code, body := decodeResponse(t, w)
|
||||
require.Equal(t, http.StatusOK, code)
|
||||
assert.Equal(t, "lighthouse", body["name"])
|
||||
assert.Equal(t, []any{"10.0.0.1"}, body["vpnAddrs"])
|
||||
|
||||
// no cert state available should be an error, not a panic
|
||||
h.pki = nil
|
||||
w = httptest.NewRecorder()
|
||||
h.handleSelf(w, r)
|
||||
code, body = decodeResponse(t, w)
|
||||
assert.Equal(t, http.StatusInternalServerError, code)
|
||||
assert.NotEmpty(t, body["error"])
|
||||
}
|
||||
|
||||
func unixHTTPClient(path string) *http.Client {
|
||||
return &http.Client{
|
||||
Timeout: time.Second,
|
||||
Transport: &http.Transport{
|
||||
DialContext: func(ctx context.Context, _, _ string) (net.Conn, error) {
|
||||
return (&net.Dialer{}).DialContext(ctx, "unix", path)
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// waitForServe polls until a GET /v1/self through client succeeds
|
||||
func waitForServe(t *testing.T, client *http.Client) {
|
||||
t.Helper()
|
||||
waitFor(t, func() bool {
|
||||
resp, err := client.Get("http://hostquery/v1/self")
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
resp.Body.Close()
|
||||
return resp.StatusCode == http.StatusOK
|
||||
})
|
||||
}
|
||||
|
||||
func skipIfNoUnixSockets(t *testing.T) {
|
||||
t.Helper()
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("unix socket tests are not supported on windows CI")
|
||||
}
|
||||
}
|
||||
|
||||
func TestInfoAPIServer_unixLifecycle(t *testing.T) {
|
||||
skipIfNoUnixSockets(t)
|
||||
h, c := newTestInfoAPIServer(t)
|
||||
h.pki = newTestPKI(t, "self", []netip.Addr{netip.MustParseAddr("10.0.0.1")})
|
||||
|
||||
sock := filepath.Join(t.TempDir(), "hq.sock")
|
||||
setInfoAPIConfig(c, true, "unix://"+sock, "")
|
||||
require.NoError(t, h.reload(c, true))
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
h.Start()
|
||||
close(done)
|
||||
}()
|
||||
|
||||
client := unixHTTPClient(sock)
|
||||
waitForServe(t, client)
|
||||
|
||||
fi, err := os.Stat(sock)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, fs.FileMode(0o600), fi.Mode().Perm())
|
||||
|
||||
resp, err := client.Get("http://hostquery/v1/host?addr=10.0.0.1")
|
||||
require.NoError(t, err)
|
||||
resp.Body.Close()
|
||||
assert.Equal(t, http.StatusOK, resp.StatusCode)
|
||||
|
||||
h.Stop()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Start did not return after Stop")
|
||||
}
|
||||
_, err = os.Stat(sock)
|
||||
assert.True(t, os.IsNotExist(err), "socket file should be unlinked on shutdown")
|
||||
}
|
||||
|
||||
func TestInfoAPIServer_staleSocket(t *testing.T) {
|
||||
skipIfNoUnixSockets(t)
|
||||
h, _ := newTestInfoAPIServer(t)
|
||||
sock := filepath.Join(t.TempDir(), "hq.sock")
|
||||
|
||||
// simulate an unclean exit, a leftover socket file with no listener
|
||||
stale, err := net.ListenUnix("unix", &net.UnixAddr{Name: sock, Net: "unix"})
|
||||
require.NoError(t, err)
|
||||
stale.SetUnlinkOnClose(false)
|
||||
require.NoError(t, stale.Close())
|
||||
_, err = os.Stat(sock)
|
||||
require.NoError(t, err, "stale socket file should exist")
|
||||
|
||||
cfg := infoAPIConfig{addr: sock, socketMode: 0o600}
|
||||
ln, err := h.listen(cfg)
|
||||
require.NoError(t, err, "a stale socket should be removed and rebound")
|
||||
require.NoError(t, ln.Close())
|
||||
}
|
||||
|
||||
func TestInfoAPIServer_existingFileNotReplaced(t *testing.T) {
|
||||
skipIfNoUnixSockets(t)
|
||||
h, _ := newTestInfoAPIServer(t)
|
||||
path := filepath.Join(t.TempDir(), "hq.sock")
|
||||
require.NoError(t, os.WriteFile(path, []byte("precious"), 0o600))
|
||||
|
||||
cfg := infoAPIConfig{addr: path, socketMode: 0o600}
|
||||
_, err := h.listen(cfg)
|
||||
require.Error(t, err, "a non-socket file at the listen path must not be replaced")
|
||||
|
||||
content, err := os.ReadFile(path)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "precious", string(content))
|
||||
}
|
||||
|
||||
func TestInfoAPIServer_reload(t *testing.T) {
|
||||
skipIfNoUnixSockets(t)
|
||||
h, c := newTestInfoAPIServer(t)
|
||||
h.pki = newTestPKI(t, "self", []netip.Addr{netip.MustParseAddr("10.0.0.1")})
|
||||
dir := t.TempDir()
|
||||
sock1 := filepath.Join(dir, "hq1.sock")
|
||||
sock2 := filepath.Join(dir, "hq2.sock")
|
||||
|
||||
// initial reload only records config, Control.Start is what launches the runtime
|
||||
setInfoAPIConfig(c, false, "unix://"+sock1, "")
|
||||
require.NoError(t, h.reload(c, true))
|
||||
assert.False(t, h.enabled.Load())
|
||||
h.runMu.Lock()
|
||||
assert.Nil(t, h.run)
|
||||
h.runMu.Unlock()
|
||||
|
||||
// enabling via reload spawns the listener
|
||||
setInfoAPIConfig(c, true, "unix://"+sock1, "")
|
||||
require.NoError(t, h.reload(c, false))
|
||||
waitForServe(t, unixHTTPClient(sock1))
|
||||
|
||||
// changing the listen path restarts on the new address
|
||||
setInfoAPIConfig(c, true, "unix://"+sock2, "")
|
||||
require.NoError(t, h.reload(c, false))
|
||||
waitForServe(t, unixHTTPClient(sock2))
|
||||
waitFor(t, func() bool {
|
||||
_, err := os.Stat(sock1)
|
||||
return os.IsNotExist(err)
|
||||
})
|
||||
|
||||
// reloading an unchanged config does not restart the runtime
|
||||
h.runMu.Lock()
|
||||
rt := h.run
|
||||
h.runMu.Unlock()
|
||||
require.NoError(t, h.reload(c, false))
|
||||
h.runMu.Lock()
|
||||
assert.Same(t, rt, h.run)
|
||||
h.runMu.Unlock()
|
||||
|
||||
// disabling stops the listener
|
||||
setInfoAPIConfig(c, false, "unix://"+sock2, "")
|
||||
require.NoError(t, h.reload(c, false))
|
||||
assert.False(t, h.enabled.Load())
|
||||
waitFor(t, func() bool {
|
||||
h.runMu.Lock()
|
||||
defer h.runMu.Unlock()
|
||||
return h.run == nil
|
||||
})
|
||||
}
|
||||
@@ -275,14 +275,6 @@ func (f *Interface) sendTo(t header.MessageType, st header.MessageSubType, ci *C
|
||||
f.sendNoMetrics(t, st, ci, hostinfo, remote, p, nb, out, 0)
|
||||
}
|
||||
|
||||
// dropExhausted records an exhaustion drop and logs once, on the crossing send, for a spent tunnel.
|
||||
func (f *Interface) dropExhausted(hostinfo *HostInfo, c uint64, msg string) {
|
||||
f.messageMetrics.TxExhausted(1)
|
||||
if c == RejectAfterMessages {
|
||||
hostinfo.logger(f.l).Error(msg)
|
||||
}
|
||||
}
|
||||
|
||||
// SendVia sends a payload through a Relay tunnel. No authentication or encryption is done
|
||||
// to the payload for the ultimate target host, making this a useful method for sending
|
||||
// handshake messages to peers through relay tunnels.
|
||||
@@ -302,14 +294,7 @@ func (f *Interface) SendVia(via *HostInfo,
|
||||
// NOTE: for goboring AESGCMTLS we need to lock because of the nonce check
|
||||
via.ConnectionState.writeLock.Lock()
|
||||
}
|
||||
c, ok := via.ConnectionState.NextMessageCounter()
|
||||
if !ok {
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
via.ConnectionState.writeLock.Unlock()
|
||||
}
|
||||
f.dropExhausted(via, c, "Dropping outbound relay packets, tunnel message counter is exhausted")
|
||||
return
|
||||
}
|
||||
c := via.ConnectionState.messageCounter.Add(1)
|
||||
|
||||
out = header.Encode(out, header.Version, header.Message, header.MessageRelay, relay.RemoteIndex, c)
|
||||
f.connectionManager.Out(via)
|
||||
@@ -376,14 +361,7 @@ func (f *Interface) sendNoMetrics(t header.MessageType, st header.MessageSubType
|
||||
// NOTE: for goboring AESGCMTLS we need to lock because of the nonce check
|
||||
ci.writeLock.Lock()
|
||||
}
|
||||
c, ok := ci.NextMessageCounter()
|
||||
if !ok {
|
||||
if noiseutil.EncryptLockNeeded {
|
||||
ci.writeLock.Unlock()
|
||||
}
|
||||
f.dropExhausted(hostinfo, c, "Dropping outbound packets, tunnel message counter is exhausted")
|
||||
return
|
||||
}
|
||||
c := ci.messageCounter.Add(1)
|
||||
|
||||
//l.WithField("trace", string(debug.Stack())).Error("out Header ", &Header{Version, t, st, 0, hostinfo.remoteIndexId, c}, p)
|
||||
out = header.Encode(out, header.Version, t, st, hostinfo.remoteIndexId, c)
|
||||
|
||||
+8
-30
@@ -2,16 +2,11 @@ package iputil
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"errors"
|
||||
|
||||
"golang.org/x/net/ipv4"
|
||||
"golang.org/x/net/ipv6"
|
||||
)
|
||||
|
||||
// ErrIPv6CouldNotFindPayload is returned when the ipv6 extension header chain is truncated before a terminal
|
||||
// upper layer protocol is reached.
|
||||
var ErrIPv6CouldNotFindPayload = errors.New("could not find payload in ipv6 packet")
|
||||
|
||||
const (
|
||||
// MaxIPv4RejectPacketSize is the largest IPv4 reject packet:
|
||||
// - 20 byte ipv4 header
|
||||
@@ -204,8 +199,8 @@ func ipv4CreateRejectTCPPacket(packet []byte, out []byte) []byte {
|
||||
}
|
||||
|
||||
func ipv6CreateRejectPacket(packet []byte, out []byte) []byte {
|
||||
proto, offset, isFragment, err := IPv6FindUpperProtocol(packet)
|
||||
if err != nil || isFragment {
|
||||
proto, offset, isFragment := ipv6FindUpperProtocol(packet)
|
||||
if isFragment {
|
||||
return nil
|
||||
}
|
||||
switch proto {
|
||||
@@ -338,18 +333,7 @@ func ipv6CreateRejectTCPPacket(packet []byte, out []byte, offset int) []byte {
|
||||
return out
|
||||
}
|
||||
|
||||
// IPv6FindUpperProtocol walks the ipv6 extension header chain and returns the upper layer protocol, the
|
||||
// offset it begins at, and whether the packet is a non-first fragment. Only the RFC 8200 and IANA extension
|
||||
// headers below are walked. Everything else, including Mobility (135), HIP (139), Shim6 (140), experimental
|
||||
// 253/254, and real upper layer protocols like SCTP or GRE, is terminal. Walking those as extension headers
|
||||
// is a firewall bypass, so they fail closed. For a non-first fragment the returned protocol is the fragmented
|
||||
// protocol and offset points at the fragment header, there is no transport header to locate. Returns
|
||||
// ErrIPv6CouldNotFindPayload if packet is smaller than an ipv6 header or the chain is truncated before a
|
||||
// terminal protocol is reached.
|
||||
func IPv6FindUpperProtocol(packet []byte) (nextHeader uint8, offset int, isFragment bool, err error) {
|
||||
if len(packet) < ipv6.HeaderLen {
|
||||
return 0, 0, false, ErrIPv6CouldNotFindPayload
|
||||
}
|
||||
func ipv6FindUpperProtocol(packet []byte) (nextHeader uint8, offset int, isFragment bool) {
|
||||
nextHeader = packet[6]
|
||||
offset = ipv6.HeaderLen
|
||||
|
||||
@@ -357,36 +341,30 @@ func IPv6FindUpperProtocol(packet []byte) (nextHeader uint8, offset int, isFragm
|
||||
switch nextHeader {
|
||||
case 0, 43, 60: // Hop-by-Hop, Routing, Destination
|
||||
if len(packet) < offset+2 {
|
||||
return nextHeader, offset, isFragment, ErrIPv6CouldNotFindPayload
|
||||
return nextHeader, offset, isFragment
|
||||
}
|
||||
nextHeader = packet[offset]
|
||||
offset += (int(packet[offset+1]) + 1) << 3
|
||||
|
||||
case 44: // Fragment
|
||||
if len(packet) < offset+8 {
|
||||
return nextHeader, offset, isFragment, ErrIPv6CouldNotFindPayload
|
||||
return nextHeader, offset, isFragment
|
||||
}
|
||||
// Non-first fragments carry no transport header, report the fragmented protocol and stop
|
||||
if packet[offset+2] != 0 || packet[offset+3]&0xf8 != 0 {
|
||||
return packet[offset], offset, true, nil
|
||||
isFragment = true
|
||||
}
|
||||
nextHeader = packet[offset]
|
||||
offset += 8
|
||||
|
||||
case 51: // AH
|
||||
if len(packet) < offset+2 {
|
||||
return nextHeader, offset, isFragment, ErrIPv6CouldNotFindPayload
|
||||
return nextHeader, offset, isFragment
|
||||
}
|
||||
nextHeader = packet[offset]
|
||||
offset += (int(packet[offset+1]) + 2) << 2
|
||||
|
||||
default:
|
||||
// A prior extension header can declare a length that advances offset past the packet. The terminal
|
||||
// protocol's header isn't actually here, so treat the chain as truncated rather than classifying it.
|
||||
if offset > len(packet) {
|
||||
return nextHeader, offset, isFragment, ErrIPv6CouldNotFindPayload
|
||||
}
|
||||
return nextHeader, offset, isFragment, nil
|
||||
return nextHeader, offset, isFragment
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,7 +6,6 @@ import (
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/net/ipv4"
|
||||
"golang.org/x/net/ipv6"
|
||||
)
|
||||
@@ -475,61 +474,3 @@ func TestCreateICMPEchoResponse_IPv6_NotICMPv6(t *testing.T) {
|
||||
result := CreateICMPEchoResponse(packet, out)
|
||||
assert.Nil(t, result)
|
||||
}
|
||||
|
||||
func Test_IPv6FindUpperProtocol(t *testing.T) {
|
||||
src := net.ParseIP("fd00::1")
|
||||
dst := net.ParseIP("fd00::2")
|
||||
|
||||
// 8 byte extension/transport stand-ins, first byte is the next header, second is the length field
|
||||
extToTCP := []byte{6, 0, 0, 0, 0, 0, 0, 0} // len 0 -> 8 bytes, next = TCP
|
||||
extToUDP := []byte{17, 0, 0, 0, 0, 0, 0, 0} // len 0 -> 8 bytes, next = UDP
|
||||
extToRouting := []byte{43, 0, 0, 0, 0, 0, 0, 0} // len 0 -> 8 bytes, next = Routing
|
||||
ahToUDP := []byte{17, 0, 0, 0, 0, 0, 0, 0} // AH len 0 -> (0+2)<<2 = 8 bytes, next = UDP
|
||||
firstFragToUDP := []byte{17, 0, 0, 1, 0, 0, 0, 1} // frag offset 0, M=1, next = UDP
|
||||
nonFirstFrag := []byte{17, 0, 0, 9, 0, 0, 0, 1} // frag offset non-zero, next = UDP
|
||||
transport := []byte{0, 80, 1, 187, 0, 0, 0, 0} // stand-in bytes, IPv6FindUpperProtocol never reads ports
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
nextHeader uint8
|
||||
payload []byte
|
||||
wantProto uint8
|
||||
wantOffset int
|
||||
wantFragment bool
|
||||
wantErr error
|
||||
}{
|
||||
{"plain udp", 17, transport, 17, ipv6.HeaderLen, false, nil},
|
||||
{"hop-by-hop then tcp", 0, append(extToTCP, transport...), 6, ipv6.HeaderLen + 8, false, nil},
|
||||
{"routing then tcp", 43, append(extToTCP, transport...), 6, ipv6.HeaderLen + 8, false, nil},
|
||||
{"destination then udp", 60, append(extToUDP, transport...), 17, ipv6.HeaderLen + 8, false, nil},
|
||||
{"hop-by-hop, routing, then tcp", 0, append(append(extToRouting, extToTCP...), transport...), 6, ipv6.HeaderLen + 16, false, nil},
|
||||
{"ah then udp", 51, append(ahToUDP, transport...), 17, ipv6.HeaderLen + 8, false, nil},
|
||||
{"first fragment walks to transport", 44, append(firstFragToUDP, transport...), 17, ipv6.HeaderLen + 8, false, nil},
|
||||
{"non-first fragment stops", 44, append(nonFirstFrag, transport...), 17, ipv6.HeaderLen, true, nil},
|
||||
{"unknown protocol is terminal", 132, transport, 132, ipv6.HeaderLen, false, nil}, // SCTP
|
||||
{"truncated extension header", 0, nil, 0, ipv6.HeaderLen, false, ErrIPv6CouldNotFindPayload},
|
||||
// Destination Options with a declared length (255+1)*8 = 2048 that runs past the 48 byte buffer, next = SCTP
|
||||
{"extension length past buffer", 60, []byte{132, 255, 0, 0, 0, 0, 0, 0}, 132, ipv6.HeaderLen + 2048, false, ErrIPv6CouldNotFindPayload},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
packet := makeIPv6Packet(src, dst, tt.nextHeader, tt.payload)
|
||||
proto, offset, isFragment, err := IPv6FindUpperProtocol(packet)
|
||||
if tt.wantErr != nil {
|
||||
assert.ErrorIs(t, err, tt.wantErr)
|
||||
return
|
||||
}
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, tt.wantProto, proto)
|
||||
assert.Equal(t, tt.wantOffset, offset)
|
||||
assert.Equal(t, tt.wantFragment, isFragment)
|
||||
})
|
||||
}
|
||||
|
||||
// A packet smaller than an ipv6 header must error rather than panic reading byte 6
|
||||
t.Run("shorter than ipv6 header", func(t *testing.T) {
|
||||
_, _, _, err := IPv6FindUpperProtocol(make([]byte, 6))
|
||||
assert.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||
})
|
||||
}
|
||||
|
||||
@@ -11,7 +11,6 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/slackhq/nebula/config"
|
||||
"github.com/slackhq/nebula/noiseutil"
|
||||
"github.com/slackhq/nebula/overlay"
|
||||
"github.com/slackhq/nebula/sshd"
|
||||
"github.com/slackhq/nebula/udp"
|
||||
@@ -21,12 +20,6 @@ import (
|
||||
|
||||
type m = map[string]any
|
||||
|
||||
// maxRoutines caps routines below the RejectHeadroom nonce gap so concurrent senders can't race the counter past wrap.
|
||||
const maxRoutines = 1 << 16
|
||||
|
||||
// The reject headroom must exceed every sender that can be mid-reservation at once, about two per routine.
|
||||
const _ = noiseutil.RejectHeadroom - 4*maxRoutines
|
||||
|
||||
func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, deviceFactory overlay.DeviceFactory) (retcon *Control, reterr error) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
// Automatically cancel the context if Main returns an error, to signal all created goroutines to quit.
|
||||
@@ -88,6 +81,9 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
if routines < 1 {
|
||||
routines = 1
|
||||
}
|
||||
if routines > 1 {
|
||||
l.Info("Using multiple routines", "routines", routines)
|
||||
}
|
||||
} else {
|
||||
// deprecated and undocumented
|
||||
tunQueues := c.GetInt("tun.routines", 1)
|
||||
@@ -97,12 +93,6 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
l.Warn("Setting tun.routines and listen.routines is deprecated. Use `routines` instead", "routines", routines)
|
||||
}
|
||||
}
|
||||
if routines > maxRoutines {
|
||||
l.Warn("Using multiple routines", "routines", maxRoutines, "clamped", true, "requestedRoutines", routines)
|
||||
routines = maxRoutines
|
||||
} else if routines > 1 {
|
||||
l.Info("Using multiple routines", "routines", routines)
|
||||
}
|
||||
|
||||
// EXPERIMENTAL
|
||||
// Intentionally not documented yet while we do more testing and determine
|
||||
@@ -270,6 +260,8 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
return nil, util.ContextualizeIfNeeded("Failed to start stats emitter", err)
|
||||
}
|
||||
|
||||
infoAPI := newInfoAPIServerFromConfig(ctx, l, pki, hostMap, c)
|
||||
|
||||
if configTest {
|
||||
return nil, nil
|
||||
}
|
||||
@@ -289,6 +281,7 @@ func Main(c *config.C, configTest bool, buildVersion string, l *slog.Logger, dev
|
||||
sshStart: sshStart,
|
||||
statsStart: stats.Start,
|
||||
dnsStart: ds.Start,
|
||||
infoAPIStart: infoAPI.Start,
|
||||
lighthouseStart: lightHouse.StartUpdateWorker,
|
||||
networkChangeStart: networkChanges.Start,
|
||||
connectionManagerStart: connManager.Start,
|
||||
|
||||
+4
-13
@@ -14,8 +14,7 @@ type MessageMetrics struct {
|
||||
rxUnknown metrics.Counter
|
||||
txUnknown metrics.Counter
|
||||
|
||||
rxInvalid metrics.Counter
|
||||
txExhausted metrics.Counter
|
||||
rxInvalid metrics.Counter
|
||||
}
|
||||
|
||||
func (m *MessageMetrics) Rx(t header.MessageType, s header.MessageSubType, i int64) {
|
||||
@@ -42,13 +41,6 @@ func (m *MessageMetrics) RxInvalid(i int64) {
|
||||
}
|
||||
}
|
||||
|
||||
// TxExhausted counts outbound packets dropped because the tunnel's message counter is spent.
|
||||
func (m *MessageMetrics) TxExhausted(i int64) {
|
||||
if m != nil && m.txExhausted != nil {
|
||||
m.txExhausted.Inc(i)
|
||||
}
|
||||
}
|
||||
|
||||
func newMessageMetrics() *MessageMetrics {
|
||||
gen := func(t string) [][]metrics.Counter {
|
||||
return [][]metrics.Counter{
|
||||
@@ -69,10 +61,9 @@ func newMessageMetrics() *MessageMetrics {
|
||||
rx: gen("rx"),
|
||||
tx: gen("tx"),
|
||||
|
||||
rxUnknown: metrics.GetOrRegisterCounter("messages.rx.other", nil),
|
||||
txUnknown: metrics.GetOrRegisterCounter("messages.tx.other", nil),
|
||||
rxInvalid: metrics.GetOrRegisterCounter("messages.rx.invalid", nil),
|
||||
txExhausted: metrics.GetOrRegisterCounter("messages.tx.exhausted", nil),
|
||||
rxUnknown: metrics.GetOrRegisterCounter("messages.rx.other", nil),
|
||||
txUnknown: metrics.GetOrRegisterCounter("messages.tx.other", nil),
|
||||
rxInvalid: metrics.GetOrRegisterCounter("messages.rx.invalid", nil),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -25,9 +25,6 @@ func (s *CipherStateAESGCM) EncryptDanger(out, ad, plaintext []byte, n uint64, n
|
||||
if s == nil {
|
||||
return nil, errors.New("no cipher state available to encrypt")
|
||||
}
|
||||
if n >= RejectAfterMessages {
|
||||
return nil, ErrMessageCounterExhausted
|
||||
}
|
||||
nb[0] = 0
|
||||
nb[1] = 0
|
||||
nb[2] = 0
|
||||
|
||||
@@ -24,9 +24,6 @@ func (s *CipherStateChaChaPoly) EncryptDanger(out, ad, plaintext []byte, n uint6
|
||||
if s == nil {
|
||||
return nil, errors.New("no cipher state available to encrypt")
|
||||
}
|
||||
if n >= RejectAfterMessages {
|
||||
return nil, ErrMessageCounterExhausted
|
||||
}
|
||||
nb[0] = 0
|
||||
nb[1] = 0
|
||||
nb[2] = 0
|
||||
|
||||
@@ -1,22 +1,11 @@
|
||||
package noiseutil
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
|
||||
"github.com/flynn/noise"
|
||||
)
|
||||
|
||||
// RejectHeadroom is the wrap gap for senders racing the counter, sized large enough for any routine count.
|
||||
const RejectHeadroom = uint64(1) << 40
|
||||
|
||||
// RejectAfterMessages is the nonce ceiling: encrypting stops RejectHeadroom short of the wrap.
|
||||
const RejectAfterMessages = math.MaxUint64 - RejectHeadroom
|
||||
|
||||
// ErrMessageCounterExhausted is returned by EncryptDanger once the nonce reaches RejectAfterMessages.
|
||||
var ErrMessageCounterExhausted = errors.New("message counter exhausted")
|
||||
|
||||
// CipherState is the post-handshake AEAD cipher used for the data plane.
|
||||
// Each supported cipher has its own concrete implementation in this package with the nonce endianness hardcoded,
|
||||
// so the encrypt/decrypt fast path avoids interface dispatch on the byte order.
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
package noiseutil
|
||||
|
||||
import (
|
||||
"math"
|
||||
"testing"
|
||||
|
||||
"github.com/flynn/noise"
|
||||
@@ -90,24 +89,6 @@ func roundtrip(t *testing.T, enc, dec CipherState) {
|
||||
assert.Equal(t, 16, enc.Overhead())
|
||||
}
|
||||
|
||||
func TestEncryptRejectsExhaustedCounter(t *testing.T) {
|
||||
// Pin the headroom below the uint64 wrap so a typo can't silently move the ceiling.
|
||||
require.Equal(t, uint64(1)<<40, RejectHeadroom)
|
||||
require.Equal(t, math.MaxUint64-RejectHeadroom, RejectAfterMessages)
|
||||
|
||||
encA, _ := buildCipherStates(t, CipherAESGCM)
|
||||
encC, _ := buildCipherStates(t, noise.CipherChaChaPoly)
|
||||
nb := make([]byte, 12)
|
||||
|
||||
for _, cs := range []CipherState{NewCipherStateAESGCM(encA), NewCipherStateChaChaPoly(encC)} {
|
||||
_, err := cs.EncryptDanger(nil, nil, []byte("x"), RejectAfterMessages-1, nb)
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = cs.EncryptDanger(nil, nil, []byte("x"), RejectAfterMessages, nb)
|
||||
require.ErrorIs(t, err, ErrMessageCounterExhausted)
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkCipherStateEncryptAESGCM(b *testing.B) {
|
||||
enc, _ := buildCipherStatesB(b, CipherAESGCM)
|
||||
benchEncryptCipherState(b, NewCipherState(enc, CipherAESGCM))
|
||||
|
||||
+85
-43
@@ -13,7 +13,6 @@ import (
|
||||
|
||||
"github.com/slackhq/nebula/firewall"
|
||||
"github.com/slackhq/nebula/header"
|
||||
"github.com/slackhq/nebula/iputil"
|
||||
"golang.org/x/net/ipv4"
|
||||
)
|
||||
|
||||
@@ -300,6 +299,7 @@ var (
|
||||
ErrIPv4InvalidHeaderLength = errors.New("invalid ipv4 header length")
|
||||
ErrIPv4PacketTooShort = errors.New("ipv4 packet is too short")
|
||||
ErrIPv6PacketTooShort = errors.New("ipv6 packet is too short")
|
||||
ErrIPv6CouldNotFindPayload = errors.New("could not find payload in ipv6 packet")
|
||||
)
|
||||
|
||||
// newPacket validates and parses the interesting bits for the firewall out of the ip and sub protocol headers
|
||||
@@ -332,59 +332,101 @@ func parseV6(data []byte, incoming bool, fp *firewall.Packet) error {
|
||||
fp.RemoteAddr, _ = netip.AddrFromSlice(data[24:40])
|
||||
}
|
||||
|
||||
// Walk the extension header chain to the upper layer protocol. iputil.IPv6FindUpperProtocol is the single
|
||||
// source of truth for which headers are extension headers, so this stays in lockstep with the reject path
|
||||
// and cannot drift into misreading an unknown protocol (SCTP, GRE, etc.) as a forged transport.
|
||||
proto, offset, isFragment, err := iputil.IPv6FindUpperProtocol(data)
|
||||
if err != nil {
|
||||
return ErrIPv6PacketTooShort
|
||||
}
|
||||
|
||||
fp.Protocol = proto
|
||||
fp.Fragment = isFragment
|
||||
if isFragment {
|
||||
// Non-first fragments carry no transport header, so we have no ports to read
|
||||
fp.RemotePort = 0
|
||||
fp.LocalPort = 0
|
||||
return nil
|
||||
}
|
||||
|
||||
switch layers.IPProtocol(proto) {
|
||||
case layers.IPProtocolICMPv6:
|
||||
// An ICMPv6 message is at least type, code and checksum, 4 bytes. Only echo carries more than we read.
|
||||
if dataLen < offset+4 {
|
||||
return ErrIPv6PacketTooShort
|
||||
protoAt := 6 // NextHeader is at 6 bytes into the ipv6 header
|
||||
offset := ipv6.HeaderLen // Start at the end of the ipv6 header
|
||||
next := 0
|
||||
for {
|
||||
if protoAt >= dataLen {
|
||||
break
|
||||
}
|
||||
fp.LocalPort = 0 //incoming vs outgoing doesn't matter for icmpv6
|
||||
switch data[offset] { //icmp type
|
||||
case layers.ICMPv6TypeEchoRequest, layers.ICMPv6TypeEchoReply:
|
||||
proto := layers.IPProtocol(data[protoAt])
|
||||
|
||||
switch proto {
|
||||
case layers.IPProtocolESP, layers.IPProtocolNoNextHeader:
|
||||
fp.Protocol = uint8(proto)
|
||||
fp.RemotePort = 0
|
||||
fp.LocalPort = 0
|
||||
fp.Fragment = false
|
||||
return nil
|
||||
|
||||
case layers.IPProtocolICMPv6:
|
||||
if dataLen < offset+6 {
|
||||
return ErrIPv6PacketTooShort
|
||||
}
|
||||
fp.RemotePort = binary.BigEndian.Uint16(data[offset+4 : offset+6]) //identifier
|
||||
fp.Protocol = uint8(proto)
|
||||
fp.LocalPort = 0 //incoming vs outgoing doesn't matter for icmpv6
|
||||
icmptype := data[offset+1]
|
||||
switch icmptype {
|
||||
case layers.ICMPv6TypeEchoRequest, layers.ICMPv6TypeEchoReply:
|
||||
fp.RemotePort = binary.BigEndian.Uint16(data[offset+4 : offset+6]) //identifier
|
||||
default:
|
||||
fp.RemotePort = 0
|
||||
}
|
||||
fp.Fragment = false
|
||||
return nil
|
||||
|
||||
case layers.IPProtocolTCP, layers.IPProtocolUDP:
|
||||
if dataLen < offset+4 {
|
||||
return ErrIPv6PacketTooShort
|
||||
}
|
||||
|
||||
fp.Protocol = uint8(proto)
|
||||
if incoming {
|
||||
fp.RemotePort = binary.BigEndian.Uint16(data[offset : offset+2])
|
||||
fp.LocalPort = binary.BigEndian.Uint16(data[offset+2 : offset+4])
|
||||
} else {
|
||||
fp.LocalPort = binary.BigEndian.Uint16(data[offset : offset+2])
|
||||
fp.RemotePort = binary.BigEndian.Uint16(data[offset+2 : offset+4])
|
||||
}
|
||||
|
||||
fp.Fragment = false
|
||||
return nil
|
||||
|
||||
case layers.IPProtocolIPv6Fragment:
|
||||
// Fragment header is 8 bytes, need at least offset+4 to read the offset field
|
||||
if dataLen < offset+8 {
|
||||
return ErrIPv6PacketTooShort
|
||||
}
|
||||
|
||||
// Check if this is the first fragment
|
||||
fragmentOffset := binary.BigEndian.Uint16(data[offset+2:offset+4]) &^ uint16(0x7) // Remove the reserved and M flag bits
|
||||
if fragmentOffset != 0 {
|
||||
// Non-first fragment, use what we have now and stop processing
|
||||
fp.Protocol = data[offset]
|
||||
fp.Fragment = true
|
||||
fp.RemotePort = 0
|
||||
fp.LocalPort = 0
|
||||
return nil
|
||||
}
|
||||
|
||||
// The next loop should be the transport layer since we are the first fragment
|
||||
next = 8 // Fragment headers are always 8 bytes
|
||||
|
||||
case layers.IPProtocolAH:
|
||||
// Auth headers, used by IPSec, have a different meaning for header length
|
||||
if dataLen <= offset+1 {
|
||||
break
|
||||
}
|
||||
next = (int(data[offset+1]) + 2) << 2
|
||||
|
||||
default:
|
||||
fp.RemotePort = 0
|
||||
// Normal ipv6 header length processing
|
||||
if dataLen <= offset+1 {
|
||||
break
|
||||
}
|
||||
next = (int(data[offset+1]) + 1) << 3
|
||||
}
|
||||
|
||||
case layers.IPProtocolTCP, layers.IPProtocolUDP:
|
||||
if dataLen < offset+4 {
|
||||
return ErrIPv6PacketTooShort
|
||||
}
|
||||
if incoming {
|
||||
fp.RemotePort = binary.BigEndian.Uint16(data[offset : offset+2])
|
||||
fp.LocalPort = binary.BigEndian.Uint16(data[offset+2 : offset+4])
|
||||
} else {
|
||||
fp.LocalPort = binary.BigEndian.Uint16(data[offset : offset+2])
|
||||
fp.RemotePort = binary.BigEndian.Uint16(data[offset+2 : offset+4])
|
||||
if next <= 0 {
|
||||
// Safety check, each ipv6 header has to be at least 8 bytes
|
||||
next = 8
|
||||
}
|
||||
|
||||
default:
|
||||
// don't set ports for protocols Nebula doesn't inspect
|
||||
fp.RemotePort = 0
|
||||
fp.LocalPort = 0
|
||||
protoAt = offset
|
||||
offset = offset + next
|
||||
}
|
||||
|
||||
return nil
|
||||
return ErrIPv6CouldNotFindPayload
|
||||
}
|
||||
|
||||
func parseV4(data []byte, incoming bool, fp *firewall.Packet) error {
|
||||
|
||||
+9
-89
@@ -14,7 +14,6 @@ import (
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"golang.org/x/net/ipv4"
|
||||
"golang.org/x/net/ipv6"
|
||||
)
|
||||
|
||||
func Test_newPacket(t *testing.T) {
|
||||
@@ -116,12 +115,12 @@ func Test_newPacket_v6(t *testing.T) {
|
||||
require.NoError(t, err)
|
||||
|
||||
err = newPacket(buffer.Bytes(), true, p)
|
||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||
|
||||
// A v6 packet with a hop-by-hop extension
|
||||
// ICMPv6 Payload (Echo Request)
|
||||
icmpLayer := layers.ICMPv6{
|
||||
TypeCode: layers.CreateICMPv6TypeCode(layers.ICMPv6TypeEchoRequest, 0),
|
||||
TypeCode: layers.ICMPv6TypeEchoRequest,
|
||||
}
|
||||
// Hop-by-Hop Extension Header
|
||||
hopOption := layers.IPv6HopByHopOption{}
|
||||
@@ -150,12 +149,12 @@ func Test_newPacket_v6(t *testing.T) {
|
||||
// A full IPv6 header and 1 byte in the first extension, but missing
|
||||
// the length byte.
|
||||
err = newPacket(buffer.Bytes()[:41], true, p)
|
||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||
|
||||
// A full IPv6 header plus 1 full extension, but only 1 byte of the
|
||||
// next layer, missing length byte
|
||||
err = newPacket(buffer.Bytes()[:49], true, p)
|
||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||
err = nil
|
||||
|
||||
// A good ICMP packet
|
||||
@@ -168,7 +167,7 @@ func Test_newPacket_v6(t *testing.T) {
|
||||
}
|
||||
|
||||
icmp := layers.ICMPv6{
|
||||
TypeCode: layers.CreateICMPv6TypeCode(layers.ICMPv6TypeEchoRequest, 0),
|
||||
TypeCode: layers.ICMPv6TypeEchoRequest,
|
||||
Checksum: 0x1234,
|
||||
}
|
||||
|
||||
@@ -190,18 +189,6 @@ func Test_newPacket_v6(t *testing.T) {
|
||||
assert.Equal(t, uint16(0), p.LocalPort)
|
||||
assert.False(t, p.Fragment)
|
||||
|
||||
// A minimal 4 byte non-echo ICMPv6 message (type, code, checksum), no identifier to read
|
||||
icmpMin := make([]byte, ipv6.HeaderLen+4)
|
||||
copy(icmpMin, buffer.Bytes()[:ipv6.HeaderLen])
|
||||
icmpMin[6] = byte(layers.IPProtocolICMPv6)
|
||||
icmpMin[ipv6.HeaderLen] = 1 // type 1, destination unreachable, not echo
|
||||
err = newPacket(icmpMin, true, p)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint8(layers.IPProtocolICMPv6), p.Protocol)
|
||||
assert.Equal(t, uint16(0), p.RemotePort)
|
||||
assert.Equal(t, uint16(0), p.LocalPort)
|
||||
assert.False(t, p.Fragment)
|
||||
|
||||
// A good ESP packet
|
||||
b := buffer.Bytes()
|
||||
b[6] = byte(layers.IPProtocolESP)
|
||||
@@ -226,15 +213,11 @@ func Test_newPacket_v6(t *testing.T) {
|
||||
assert.Equal(t, uint16(0), p.LocalPort)
|
||||
assert.False(t, p.Fragment)
|
||||
|
||||
// An unknown protocol packet, we don't dissect it so we fail closed on its true protocol with no ports
|
||||
// An unknown protocol packet
|
||||
b = buffer.Bytes()
|
||||
b[6] = 255 // 255 is a reserved protocol number
|
||||
err = newPacket(b, true, p)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint8(255), p.Protocol)
|
||||
assert.Equal(t, uint16(0), p.RemotePort)
|
||||
assert.Equal(t, uint16(0), p.LocalPort)
|
||||
assert.False(t, p.Fragment)
|
||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||
|
||||
// A good UDP packet
|
||||
ip = layers.IPv6{
|
||||
@@ -351,14 +334,14 @@ func Test_newPacket_v6(t *testing.T) {
|
||||
assert.Equal(t, uint16(22), p.LocalPort)
|
||||
assert.False(t, p.Fragment)
|
||||
|
||||
// Ensure buffer bounds checking during processing, a truncated AH header can't reach the payload
|
||||
// Ensure buffer bounds checking during processing
|
||||
err = newPacket(b[:41], true, p)
|
||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||
|
||||
// Invalid AH header
|
||||
b = buffer.Bytes()
|
||||
err = newPacket(b, true, p)
|
||||
require.ErrorIs(t, err, ErrIPv6PacketTooShort)
|
||||
require.ErrorIs(t, err, ErrIPv6CouldNotFindPayload)
|
||||
}
|
||||
|
||||
func Test_newPacket_ipv6Fragment(t *testing.T) {
|
||||
@@ -692,66 +675,3 @@ func Test_newPacket_v6ExtHeaderOverflow(t *testing.T) {
|
||||
// the host delivers to, not the forged 443 at the overflowed offset.
|
||||
assert.Equal(t, uint16(22), p.LocalPort, "firewall must parse the real transport header, not the overflowed offset")
|
||||
}
|
||||
|
||||
// Test_newPacket_v6ExtHeaderPastBuffer is a regression test for an extension header whose declared length
|
||||
// advances the walk past the end of the packet. The upper layer protocol's header isn't actually present,
|
||||
// so parseV6 must drop the packet rather than classify it as the terminal protocol with no ports.
|
||||
func Test_newPacket_v6ExtHeaderPastBuffer(t *testing.T) {
|
||||
p := &firewall.Packet{}
|
||||
|
||||
pkt := make([]byte, 48)
|
||||
pkt[0] = 0x60
|
||||
pkt[6] = byte(layers.IPProtocolIPv6Destination) // Destination Options
|
||||
pkt[7] = 64 // hop limit
|
||||
pkt[40] = byte(layers.IPProtocolSCTP) // Dest Options next header = SCTP
|
||||
pkt[41] = 255 // declared length (255+1)*8 = 2048, past the 48 byte buffer
|
||||
|
||||
require.ErrorIs(t, newPacket(pkt, true, p), ErrIPv6PacketTooShort)
|
||||
}
|
||||
|
||||
// Test_newPacket_v6ExtHeaderConfusion is a regression test for parseV6 walking any unrecognized
|
||||
// Next Header as if it were an ipv6 extension header. A real upper layer protocol Nebula doesn't
|
||||
// dissect (SCTP here) is not walkable, so applying the (len+1)*8 formula marched into the SCTP
|
||||
// payload and landed on a byte that looked like UDP, forging a protocol/port pair the firewall
|
||||
// would trust while the host delivered the real SCTP datagram. The fix fails closed: the packet
|
||||
// is classified as its true protocol with no ports, so it only matches an `any` rule.
|
||||
func Test_newPacket_v6ExtHeaderConfusion(t *testing.T) {
|
||||
p := &firewall.Packet{}
|
||||
|
||||
pkt := make([]byte, 52)
|
||||
pkt[0] = 0x60 // version 6
|
||||
pkt[6] = byte(layers.IPProtocolSCTP) // NextHeader = SCTP, a real protocol, not an extension header
|
||||
pkt[7] = 64 // hop limit
|
||||
|
||||
// Real SCTP header at offset 40. Pre-fix parseV6 walked SCTP as an extension header: byte 41 (0x00, the
|
||||
// low byte of the src port below) was read as the header length, giving next=(0+1)*8=8, which landed the
|
||||
// walk on byte 40 (0x11), misread as NextHeader=UDP, then bytes 48-51 as ports.
|
||||
binary.BigEndian.PutUint16(pkt[40:42], 0x1100) // SCTP src port; byte 40=0x11, byte 41=0x00
|
||||
binary.BigEndian.PutUint16(pkt[42:44], 445) // SCTP dst port, never read by parseV6
|
||||
binary.BigEndian.PutUint16(pkt[48:50], 53) // SCTP checksum bytes, pre-fix forged RemotePort
|
||||
binary.BigEndian.PutUint16(pkt[50:52], 53) // pre-fix forged LocalPort
|
||||
|
||||
require.NoError(t, newPacket(pkt, true, p))
|
||||
assert.Equal(t, uint8(layers.IPProtocolSCTP), p.Protocol, "must classify as the true protocol, not the forged UDP")
|
||||
assert.Equal(t, uint16(0), p.RemotePort)
|
||||
assert.Equal(t, uint16(0), p.LocalPort)
|
||||
assert.False(t, p.Fragment)
|
||||
|
||||
// Same confusion, but the unknown protocol sits after a real extension header. The HopByHop is walked
|
||||
// correctly, then SCTP must still fail closed instead of being walked into its own payload. Protocol is
|
||||
// the only assertion that discriminates the fix here, a regression that walked SCTP would misclassify it.
|
||||
chained := make([]byte, 60)
|
||||
chained[0] = 0x60 // version 6
|
||||
chained[6] = byte(layers.IPProtocolIPv6HopByHop) // NextHeader = HopByHop extension
|
||||
chained[7] = 64 // hop limit
|
||||
chained[40] = byte(layers.IPProtocolSCTP) // HopByHop NextHeader = SCTP
|
||||
chained[41] = 0 // HopByHop length 0 -> 8 bytes, SCTP begins at offset 48
|
||||
binary.BigEndian.PutUint16(chained[48:50], 0x1100) // SCTP src port, pre-fix forged NextHeader/length bait
|
||||
binary.BigEndian.PutUint16(chained[50:52], 445) // SCTP dst port, never read by parseV6
|
||||
|
||||
require.NoError(t, newPacket(chained, true, p))
|
||||
assert.Equal(t, uint8(layers.IPProtocolSCTP), p.Protocol, "must fail closed on the unknown protocol after the extension header")
|
||||
assert.Equal(t, uint16(0), p.RemotePort)
|
||||
assert.Equal(t, uint16(0), p.LocalPort)
|
||||
assert.False(t, p.Fragment)
|
||||
}
|
||||
|
||||
+1
-11
@@ -5,7 +5,6 @@ package overlay
|
||||
|
||||
import (
|
||||
"encoding/binary"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log/slog"
|
||||
@@ -484,16 +483,7 @@ func (t *tun) addIPs(link netlink.Link) error {
|
||||
//iterate over remainder, remove whoever shouldn't be there
|
||||
al, err := netlink.AddrList(link, netlink.FAMILY_ALL)
|
||||
if err != nil {
|
||||
//RTM_GETADDR dumps the whole system, so any concurrent address change
|
||||
//interrupts it - including the kernel's async tentative->preferred
|
||||
//flip of an IPv6 address the AddrReplace calls above just added,
|
||||
//which makes this a race against our own setup. Partial results are
|
||||
//still returned; the worst case is a stale address surviving until
|
||||
//the next config reload, which beats failing startup over it.
|
||||
if !errors.Is(err, netlink.ErrDumpInterrupted) {
|
||||
return fmt.Errorf("failed to get tun address list: %s", err)
|
||||
}
|
||||
t.l.Warn("tun address list dump was interrupted, stale addresses may remain")
|
||||
return fmt.Errorf("failed to get tun address list: %s", err)
|
||||
}
|
||||
|
||||
for i := range al {
|
||||
|
||||
@@ -13,6 +13,7 @@ import (
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
graphite "github.com/cyberdelia/go-metrics-graphite"
|
||||
mp "github.com/nbrownus/go-metrics-prometheus"
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/prometheus/client_golang/prometheus/promhttp"
|
||||
@@ -252,7 +253,7 @@ func (s *statsServer) buildRuntime(cfg statsConfig) ([]func(), *http.Server) {
|
||||
// loadStatsConfig already resolved and validated the address; re-parse
|
||||
// the resolved form (no DNS lookup) to get a *net.TCPAddr.
|
||||
addr, _ := net.ResolveTCPAddr(cfg.graphite.protocol, cfg.graphite.resolvedAddr)
|
||||
gcfg := graphiteConfigExport{
|
||||
gcfg := graphite.Config{
|
||||
Addr: addr,
|
||||
Registry: metrics.DefaultRegistry,
|
||||
FlushInterval: cfg.interval,
|
||||
@@ -261,7 +262,7 @@ func (s *statsServer) buildRuntime(cfg statsConfig) ([]func(), *http.Server) {
|
||||
Percentiles: []float64{0.5, 0.75, 0.95, 0.99, 0.999},
|
||||
}
|
||||
captureFns = append(captureFns, func() {
|
||||
if err := graphiteOnce(gcfg); err != nil {
|
||||
if err := graphite.Once(gcfg); err != nil {
|
||||
s.l.Error("Graphite export failed", "error", err)
|
||||
}
|
||||
})
|
||||
|
||||
+1
-1
@@ -371,7 +371,7 @@ func waitForListening(t *testing.T, addr string) {
|
||||
})
|
||||
}
|
||||
|
||||
// graphiteSink is a minimal TCP accept-and-discard server so graphiteOnce
|
||||
// graphiteSink is a minimal TCP accept-and-discard server so graphite.Once
|
||||
// calls in tests don't spam error logs or wedge on connection refused.
|
||||
type graphiteSink struct {
|
||||
ln net.Listener
|
||||
|
||||
Reference in New Issue
Block a user