Files
tailscale/ipn/ipnlocal/node_backend_test.go
Brad FitzpatrickandBrad Fitzpatrick ff1c7ef23c ipn/ipnlocal,net/routemanager: keep a routemanager.RouteManager updated per node
Give nodeBackend a RouteManager and keep it in sync as routing
inputs change: full netmaps resync the whole peer set (removals plus
no-op-cheap upserts), incremental netmap deltas mirror their peer
upserts and removes into the same mutation batch, and
authReconfigLocked pushes the routing-relevant prefs (exit node,
subnet route acceptance, OneCGNAT) after resolving the exit node's
stable ID to its current numeric node ID.

A selected exit node that doesn't resolve to a current peer (a
nonexistent node, or MDM's "auto:any" placeholder awaiting
resolution) is not the same as no exit node: per the long-standing
ipn.Prefs.ExitNodeID contract, it blackholes internet traffic rather
than letting it escape to the local network. RouteManager's Prefs
gains an ExitNodeSelected bit so its OS route set keeps the default
routes in that case, with no outbound peer to carry them, matching
what routerConfigLocked does today, as pinned by TestRouterConfigExitNodeBlackhole in the previous commit.

All mutations happen with nodeBackend.mu held, satisfying the
RouteManager's serialized Begin/Commit contract.

Nothing consumes its snapshots yet; the wgengine data plane and OS
router wiring come next.

Updates #12542

Change-Id: I677b6b2c9efb8e41b3d27071bd9db73e01640d3b
Signed-off-by: Brad Fitzpatrick <[email protected]>
2026-07-13 10:20:54 -07:00

356 lines
10 KiB
Go

// Copyright (c) Tailscale Inc & contributors
// SPDX-License-Identifier: BSD-3-Clause
package ipnlocal
import (
"context"
"errors"
"maps"
"net/netip"
"slices"
"testing"
"time"
"tailscale.com/net/routecheck/peernode"
"tailscale.com/tailcfg"
"tailscale.com/tstest"
"tailscale.com/types/key"
"tailscale.com/types/netmap"
"tailscale.com/util/eventbus"
"tailscale.com/util/mak"
"tailscale.com/util/set"
)
func TestNodeBackendReadiness(t *testing.T) {
nb := newNodeBackend(t.Context(), tstest.WhileTestRunningLogger(t), eventbus.New())
// The node backend is not ready until [nodeBackend.ready] is called,
// and [nodeBackend.Wait] should fail with [context.DeadlineExceeded].
ctx, cancelCtx := context.WithTimeout(context.Background(), 100*time.Millisecond)
defer cancelCtx()
if err := nb.Wait(ctx); err != ctx.Err() {
t.Fatalf("Wait: got %v; want %v", err, ctx.Err())
}
// Start a goroutine to wait for the node backend to become ready.
waitDone := make(chan struct{})
go func() {
if err := nb.Wait(context.Background()); err != nil {
t.Errorf("Wait: got %v; want nil", err)
}
close(waitDone)
}()
// Call [nodeBackend.ready] to indicate that the node backend is now ready.
go nb.ready()
// Once the backend is called, [nodeBackend.Wait] should return immediately without error.
if err := nb.Wait(context.Background()); err != nil {
t.Fatalf("Wait: got %v; want nil", err)
}
// And any pending waiters should also be unblocked.
<-waitDone
}
func TestNodeBackendShutdown(t *testing.T) {
nb := newNodeBackend(t.Context(), tstest.WhileTestRunningLogger(t), eventbus.New())
shutdownCause := errors.New("test shutdown")
// Start a goroutine to wait for the node backend to become ready.
// This test expects it to block until the node backend shuts down
// and then return the specified shutdown cause.
waitDone := make(chan struct{})
go func() {
if err := nb.Wait(context.Background()); err != shutdownCause {
t.Errorf("Wait: got %v; want %v", err, shutdownCause)
}
close(waitDone)
}()
// Call [nodeBackend.shutdown] to indicate that the node backend is shutting down.
nb.shutdown(shutdownCause)
// Calling it again is fine, but should not change the shutdown cause.
nb.shutdown(errors.New("test shutdown again"))
// After shutdown, [nodeBackend.Wait] should return with the specified shutdown cause.
if err := nb.Wait(context.Background()); err != shutdownCause {
t.Fatalf("Wait: got %v; want %v", err, shutdownCause)
}
// The context associated with the node backend should also be cancelled
// and its cancellation cause should match the shutdown cause.
if err := nb.Context().Err(); !errors.Is(err, context.Canceled) {
t.Fatalf("Context.Err: got %v; want %v", err, context.Canceled)
}
if cause := context.Cause(nb.Context()); cause != shutdownCause {
t.Fatalf("Cause: got %v; want %v", cause, shutdownCause)
}
// And any pending waiters should also be unblocked.
<-waitDone
}
func TestNodeBackendReadyAfterShutdown(t *testing.T) {
nb := newNodeBackend(t.Context(), tstest.WhileTestRunningLogger(t), eventbus.New())
shutdownCause := errors.New("test shutdown")
nb.shutdown(shutdownCause)
nb.ready() // Calling ready after shutdown is a no-op, but should not panic, etc.
if err := nb.Wait(context.Background()); err != shutdownCause {
t.Fatalf("Wait: got %v; want %v", err, shutdownCause)
}
}
func TestNodeBackendParentContextCancellation(t *testing.T) {
ctx, cancelCtx := context.WithCancel(context.Background())
nb := newNodeBackend(ctx, tstest.WhileTestRunningLogger(t), eventbus.New())
cancelCtx()
// Cancelling the parent context should cause [nodeBackend.Wait]
// to return with [context.Canceled].
if err := nb.Wait(context.Background()); !errors.Is(err, context.Canceled) {
t.Fatalf("Wait: got %v; want %v", err, context.Canceled)
}
// And the node backend's context should also be cancelled.
if err := nb.Context().Err(); !errors.Is(err, context.Canceled) {
t.Fatalf("Context.Err: got %v; want %v", err, context.Canceled)
}
}
func TestNodeBackendConcurrentReadyAndShutdown(t *testing.T) {
nb := newNodeBackend(t.Context(), tstest.WhileTestRunningLogger(t), eventbus.New())
// Calling [nodeBackend.ready] and [nodeBackend.shutdown] concurrently
// should not cause issues, and [nodeBackend.Wait] should unblock,
// but the result of [nodeBackend.Wait] is intentionally undefined.
go nb.ready()
go nb.shutdown(errors.New("test shutdown"))
nb.Wait(context.Background())
}
func TestNodeBackendReachability(t *testing.T) {
for _, tc := range []struct {
name string
// Cap sets [tailcfg.NodeAttrClientSideReachability] on the self
// node.
//
// When disabled, the client relies on the control plane sending
// an accurate peer.Online flag. When enabled, the client
// ignores peer.Online and is forced to return true.
cap bool
// rchk sets [tailcfg.NodeAttrClientSideReachabilityRouteCheck]
// on the self node.
//
// When enabled with [tailcfg.NodeAttrClientSideReachability]
// above, the client ignores peer.Online and determines whether
// it can reach the peer node using [routecheck] reports.
rchk bool
online bool
pong peernode.Reachability
want bool
}{
{
name: "disabled/offline",
cap: false,
online: false,
want: false,
},
{
name: "disabled/online",
cap: false,
online: true,
want: true,
},
{
name: "forced/offline",
cap: true,
rchk: false,
online: false,
want: true,
},
{
name: "forced/online",
cap: true,
rchk: false,
online: true,
want: true,
},
{
name: "routecheck/offline/needs-probe",
cap: true,
rchk: true,
online: false,
pong: peernode.Unknown,
want: false,
},
{
name: "routecheck/offline/unreachable",
cap: true,
rchk: true,
online: false,
pong: peernode.Unreachable,
want: false,
},
{
name: "routecheck/offline/reachable",
cap: true,
rchk: true,
online: false,
pong: peernode.Reachable,
want: true,
},
{
name: "routecheck/online/needs-probe",
cap: true,
rchk: true,
online: true,
pong: peernode.Unknown,
want: true,
},
{
name: "routecheck/online/unreachable",
cap: true,
rchk: true,
online: true,
pong: peernode.Unreachable,
want: false,
},
{
name: "routecheck/online/reachable",
cap: true,
rchk: true,
online: true,
pong: peernode.Reachable,
want: true,
},
} {
t.Run(tc.name, func(t *testing.T) {
self := &tailcfg.Node{
ID: 1,
StableID: "stable1",
Name: "self",
}
if tc.cap {
mak.Set(&self.CapMap, tailcfg.NodeAttrClientSideReachability, nil)
}
if tc.rchk {
mak.Set(&self.CapMap, tailcfg.NodeAttrClientSideReachabilityRouteCheck, nil)
}
peer := &tailcfg.Node{
ID: 2,
StableID: "stable2",
Name: "peer",
Online: &tc.online,
}
nb := newNodeBackend(t.Context(), tstest.WhileTestRunningLogger(t), eventbus.New())
nb.netMap = &netmap.NetworkMap{
SelfNode: self.View(),
Peers: []tailcfg.NodeView{peer.View()},
// HACK: AllCaps is usually populated by Control
AllCaps: set.SetOf(slices.Collect(maps.Keys(self.CapMap))),
}
got := nb.PeerIsReachable(routecheckReport(tc.pong), peer.View())
if got != tc.want {
t.Errorf("got %v, want %v", got, tc.want)
}
})
}
}
type routecheckReport peernode.Reachability
var _ RouteCheckReport = *new(routecheckReport)
func (rp routecheckReport) IsReachable(_ tailcfg.NodeID) peernode.Reachability {
return peernode.Reachability(rp)
}
func TestNodeBackendRouteManager(t *testing.T) {
nb := newNodeBackend(t.Context(), tstest.WhileTestRunningLogger(t), eventbus.New())
mkPeer := func(id tailcfg.NodeID, stableID tailcfg.StableNodeID, addr4 string, extra ...string) tailcfg.NodeView {
n := &tailcfg.Node{
ID: id,
StableID: stableID,
Key: key.NewNode().Public(),
HomeDERP: 1, // required by the route manager's reachability filter
Addresses: []netip.Prefix{
netip.MustParsePrefix(addr4),
},
}
n.AllowedIPs = append(n.AllowedIPs, n.Addresses...)
for _, s := range extra {
n.AllowedIPs = append(n.AllowedIPs, netip.MustParsePrefix(s))
}
return n.View()
}
wantPeerFor := func(ip string, want tailcfg.NodeView) {
t.Helper()
got, ok := nb.routeMgr.Outbound().Lookup(netip.MustParseAddr(ip))
if !want.Valid() {
if ok {
t.Errorf("Outbound lookup %s = %v; want no match", ip, got)
}
return
}
if !ok || got.Key != want.Key() {
t.Errorf("Outbound lookup %s = %v, %v; want %v", ip, got, ok, want.Key())
}
}
p1 := mkPeer(1, "stable1", "100.64.0.1/32")
p2 := mkPeer(2, "stable2", "100.64.0.2/32", "0.0.0.0/0", "::/0")
// A full netmap populates the route manager.
nb.SetNetMap(&netmap.NetworkMap{Peers: []tailcfg.NodeView{p1, p2}})
wantPeerFor("100.64.0.1", p1)
wantPeerFor("100.64.0.2", p2)
wantPeerFor("8.8.8.8", tailcfg.NodeView{}) // exit node not selected
// Selecting peer 2 as the exit node resolves its stable ID and
// installs its /0 routes.
nb.updateRouteManagerPrefs(routePrefs{ExitNodeID: "stable2", ExitNodeSelected: true})
wantPeerFor("8.8.8.8", p2)
// A selected exit node that resolves to no current peer must
// blackhole internet traffic, not fall back to "no exit node":
// the default routes stay in the OS route set with no outbound
// peer to carry them.
nb.updateRouteManagerPrefs(routePrefs{ExitNodeID: "no-such-node", ExitNodeSelected: true})
wantPeerFor("8.8.8.8", tailcfg.NodeView{})
if !nb.routeMgr.OSRoutes().Get(netip.MustParsePrefix("0.0.0.0/0")) {
t.Error("unresolved exit node: OSRoutes missing 0.0.0.0/0 blackhole route")
}
nb.updateRouteManagerPrefs(routePrefs{})
wantPeerFor("8.8.8.8", tailcfg.NodeView{})
if nb.routeMgr.OSRoutes().Get(netip.MustParsePrefix("0.0.0.0/0")) {
t.Error("no exit node: OSRoutes unexpectedly contains 0.0.0.0/0")
}
// Incremental deltas: add peer 3, remove peer 1.
p3 := mkPeer(3, "stable3", "100.64.0.3/32")
if !nb.UpdateNetmapDelta([]netmap.NodeMutation{
netmap.NodeMutationUpsert{Node: p3},
netmap.MakeNodeMutationRemove(1),
}) {
t.Fatal("UpdateNetmapDelta not handled")
}
wantPeerFor("100.64.0.3", p3)
wantPeerFor("100.64.0.1", tailcfg.NodeView{})
// A full netmap that drops a peer removes it from the route manager.
nb.SetNetMap(&netmap.NetworkMap{Peers: []tailcfg.NodeView{p2}})
wantPeerFor("100.64.0.3", tailcfg.NodeView{})
wantPeerFor("100.64.0.2", p2)
}