ipn/ipnlocal, control/controlclient: process node adds/removes in constant time

For large tailnets (~50k+ nodes) with frequent peer churn (ephemeral
GitHub Actions workers etc.), tailscaled used to rebuild the full
netmap and fan it out on the IPN bus on every MapResponse that
added or removed a peer. There were two O(N) costs per delta: the
full netmap rebuild + every Notify.NetMap encode to every bus watcher.

This change tackles both:

  1. Plumb O(1) peer add/remove through the delta path. PeersChanged
     and PeersRemoved no longer prevent the delta happy path; instead,
     they mutate the per-node-backend peer map in place.

  2. Restrict ipn.Notify.NetMap emission to the platforms whose host
     GUIs still depend on it (Windows, macOS, iOS) and migrate
     in-tree consumers off it everywhere else:

     - Migrate reactive consumers (containerboot, kube agents,
       sniproxy, tsconsensus, etc.) off Notify.NetMap to the
       previously-added Notify.SelfChange signal so they no longer
       have to subscribe to the full netmap.
     - Add ipn.NotifyNoNetMap so GUI clients on "legacy-emit" platforms
       that have already migrated can opt out of the per-watcher
       NetMap encode.
     - Gate Notify.NetMap emission on the producer side by a compile-
       time GOOS check, so the supporting code is dead-code-eliminated
       on Linux and other geese where no GUI consumer needs it.

Re-running BenchmarkGiantTailnet from tstest/largetailnet, which was
added along with baseline numbers on unmodified main in ad5436af0d,
the per-delta cost (one peer add+remove pair) is now ~O(1) regardless
of tailnet size N:

    N         no-watcher (ms/op)            bus-watcher (ms/op)
              before    now     factor      before    now     factor
     10000        32   0.11       300x         166   0.13      1300x
     50000       222   0.11      2000x         865   0.13      6700x
    100000       504   0.12      4100x        1765   0.13     13400x
    250000      1551   0.12     12500x        4696   0.15     32400x

Updates #12542

Change-Id: I94e34b37331d1a8ec74c299deffadf4d061fda9e
Signed-off-by: Brad Fitzpatrick <bradfitz@tailscale.com>
This commit is contained in:
Brad Fitzpatrick
2026-05-21 09:26:19 -07:00
committed by Brad Fitzpatrick
parent 2703f91174
commit aa5da2e5f2
23 changed files with 1521 additions and 211 deletions
+72 -4
View File
@@ -20,10 +20,12 @@ import (
"tailscale.com/types/netmap"
"tailscale.com/types/persist"
"tailscale.com/types/structs"
"tailscale.com/types/views"
"tailscale.com/util/backoff"
"tailscale.com/util/clientmetric"
"tailscale.com/util/execqueue"
"tailscale.com/util/testenv"
"tailscale.com/wgengine/filter"
)
type LoginGoal struct {
@@ -479,11 +481,77 @@ func (mrs mapRoutineState) UpdateNetmapDelta(muts []netmap.NodeMutation) bool {
ctx, cancel := context.WithTimeout(c.mapCtx, 2*time.Second)
defer cancel()
var ok bool
err := c.observerQueue.RunSync(ctx, func() {
ok = ndu.UpdateNetmapDelta(muts)
ch := make(chan bool, 1)
c.observerQueue.Add(func() {
ch <- ndu.UpdateNetmapDelta(muts)
})
return err == nil && ok
select {
case ok := <-ch:
return ok
case <-ctx.Done():
return false
}
}
var (
_ PacketFilterUpdater = mapRoutineState{}
_ UserProfileUpdater = mapRoutineState{}
)
// UpdatePacketFilter implements [PacketFilterUpdater] by forwarding to
// [Auto.observer] if it implements [PacketFilterUpdater]. It returns
// false (signaling fall back to a full netmap rebuild) if the
// downstream observer doesn't implement [PacketFilterUpdater] or isn't
// in a state to accept updates.
func (mrs mapRoutineState) UpdatePacketFilter(rules views.Slice[tailcfg.FilterRule], parsed []filter.Match) bool {
c := mrs.c
c.mu.Lock()
goodState := c.loggedIn && c.inMapPoll
pfu, ok := c.observer.(PacketFilterUpdater)
c.mu.Unlock()
if !goodState || !ok {
return false
}
ctx, cancel := context.WithTimeout(c.mapCtx, 2*time.Second)
defer cancel()
ch := make(chan bool, 1)
c.observerQueue.Add(func() {
ch <- pfu.UpdatePacketFilter(rules, parsed)
})
select {
case applied := <-ch:
return applied
case <-ctx.Done():
return false
}
}
// UpdateUserProfiles implements [UserProfileUpdater] by forwarding to
// [Auto.observer] if it implements [UserProfileUpdater]. It returns
// false (signaling fall back to a full netmap rebuild) if the
// downstream observer doesn't implement [UserProfileUpdater] or isn't
// in a state to accept updates.
func (mrs mapRoutineState) UpdateUserProfiles(profiles map[tailcfg.UserID]tailcfg.UserProfileView) bool {
c := mrs.c
c.mu.Lock()
goodState := c.loggedIn && c.inMapPoll
upu, ok := c.observer.(UserProfileUpdater)
c.mu.Unlock()
if !goodState || !ok {
return false
}
ctx, cancel := context.WithTimeout(c.mapCtx, 2*time.Second)
defer cancel()
ch := make(chan bool, 1)
c.observerQueue.Add(func() {
ch <- upu.UpdateUserProfiles(profiles)
})
select {
case applied := <-ch:
return applied
case <-ctx.Done():
return false
}
}
var _ patchDiscoKeyer = mapRoutineState{}
+51
View File
@@ -56,6 +56,7 @@ import (
"tailscale.com/types/netmap"
"tailscale.com/types/persist"
"tailscale.com/types/tkatype"
"tailscale.com/types/views"
"tailscale.com/util/clientmetric"
"tailscale.com/util/eventbus"
"tailscale.com/util/singleflight"
@@ -64,6 +65,7 @@ import (
"tailscale.com/util/testenv"
"tailscale.com/util/vizerror"
"tailscale.com/util/zstdframe"
"tailscale.com/wgengine/filter"
)
// Direct is the client that connects to a tailcontrol server for a node.
@@ -226,6 +228,9 @@ type NetmapUpdater interface {
// rather than just full updates.
type NetmapDeltaUpdater interface {
// UpdateNetmapDelta is called with discrete changes to the network map.
// The mutation slice may contain [netmap.NodeMutationAdd] and
// [netmap.NodeMutationRemove] entries when peers were added or removed,
// alongside per-field patches.
//
// The ok result is whether the implementation was able to apply the
// mutations. It might return false if its internal state doesn't
@@ -234,6 +239,52 @@ type NetmapDeltaUpdater interface {
UpdateNetmapDelta([]netmap.NodeMutation) (ok bool)
}
// PacketFilterUpdater is an optional interface that can be implemented by
// NetmapUpdater implementations to receive incremental packet-filter updates
// without a full netmap rebuild.
//
// It exists because the packet filter currently changes on every peer
// addition, so a MapResponse carrying PeersChanged almost always also carries
// PacketFilter (or PacketFilters). Handling the filter narrowly keeps peer
// churn O(1) on the controlclient side.
type PacketFilterUpdater interface {
// UpdatePacketFilter is called when a MapResponse's PacketFilter (or
// PacketFilters) changed. rules is the already-merged concatenation of
// the session's named packet filter chunks; parsed is the parsed form.
//
// It returns false to signal the caller to fall back to a full
// netmap rebuild. Proxy/forwarder implementations return false when
// their downstream destination doesn't implement
// [PacketFilterUpdater]; concrete implementations return true on
// successful apply.
UpdatePacketFilter(rules views.Slice[tailcfg.FilterRule], parsed []filter.Match) bool
}
// UserProfileUpdater is an optional interface that can be implemented by
// NetmapUpdater implementations to receive incremental UserProfile updates
// without a full netmap rebuild.
//
// It exists so consumers of [ipn.Notify.UserProfiles] can be told about
// new or updated UserProfiles before (or with) the [ipn.Notify.PeersChanged]
// or [ipn.Notify.PeerChangedPatch] entry that references the corresponding
// UserID.
type UserProfileUpdater interface {
// UpdateUserProfiles is called when a MapResponse carries UserProfiles
// entries. profiles is the new/updated subset (NOT the full map);
// implementations should merge with whatever they already know.
//
// The values are [tailcfg.UserProfileView]s sharing backing memory
// with the caller's tracking map; implementations may store them
// directly without copying.
//
// It returns false to signal the caller to fall back to a full
// netmap rebuild. Proxy/forwarder implementations return false when
// their downstream destination doesn't implement
// [UserProfileUpdater]; concrete implementations return true on
// successful apply.
UpdateUserProfiles(profiles map[tailcfg.UserID]tailcfg.UserProfileView) bool
}
// patchDiscoKeyer is an optional interface that can be implemented by an [Observer] to be
// notified about node disco keys received out-of-band from control, via
// existing connection state.
+53 -2
View File
@@ -316,9 +316,11 @@ func (ms *mapSession) handleNonKeepAliveMapResponse(ctx context.Context, resp *t
}
if ms.tryHandleIncrementally(resp) {
metricMapResponseHandledIncrementally.Add(1)
ms.occasionallyPrintSummary(ms.lastNetmapSummary)
return nil
}
metricMapResponseHandledFullRebuild.Add(1)
// We have to rebuild the whole netmap (lots of garbage & work downstream of
// our UpdateFullNetmap call). This is the part we tried to avoid but
@@ -393,11 +395,49 @@ func (ms *mapSession) tryHandleIncrementally(res *tailcfg.MapResponse) bool {
if !ok {
return false
}
// If the response carries a new packet filter, the updater must
// support pushing it narrowly; otherwise fall back to a full netmap
// rebuild. PacketFilter/PacketFilters are no longer in
// mapResponseContainsNonPatchFields, so MutationsFromMapResponse will
// happily return mutations alongside a filter change — we need to
// deliver the filter separately before those mutations land.
if res.PacketFilter != nil || res.PacketFilters != nil {
pfu, ok := ms.netmapUpdater.(PacketFilterUpdater)
if !ok {
return false
}
if !pfu.UpdatePacketFilter(ms.lastPacketFilterRules, ms.lastParsedPacketFilter) {
return false
}
}
// Same shape for UserProfiles: deliver any new/updated profiles before
// the peer mutations that may reference them, so bus consumers never
// see a UserID for which a profile hasn't been published. The values
// are read from ms.lastUserProfile (just populated by
// updateStateFromResponse) so views are shared with mapSession's
// store; downstream consumers can use [UserProfileView.Equal] for
// dedup without copying.
if len(res.UserProfiles) > 0 {
upu, ok := ms.netmapUpdater.(UserProfileUpdater)
if !ok {
return false
}
profiles := make(map[tailcfg.UserID]tailcfg.UserProfileView, len(res.UserProfiles))
for _, up := range res.UserProfiles {
profiles[up.ID] = ms.lastUserProfile[up.ID]
}
if !upu.UpdateUserProfiles(profiles) {
return false
}
}
mutations, ok := netmap.MutationsFromMapResponse(res, time.Now())
if ok && len(mutations) > 0 {
if !ok {
return false
}
if len(mutations) > 0 {
return nud.UpdateNetmapDelta(mutations)
}
return ok
return true
}
// updateStats are some stats from updateStateFromResponse, primarily for
@@ -697,6 +737,17 @@ var (
patchifiedPeer = clientmetric.NewCounter("controlclient_patchified_peer")
patchifiedPeerEqual = clientmetric.NewCounter("controlclient_patchified_peer_equal")
// metricMapResponseHandledIncrementally counts non-keepalive MapResponses
// that were processed via [mapSession.tryHandleIncrementally] (i.e. the
// "fast" delta path that avoids rebuilding the full netmap).
metricMapResponseHandledIncrementally = clientmetric.NewCounter("controlclient_map_response_handled_incrementally")
// metricMapResponseHandledFullRebuild counts non-keepalive MapResponses
// that fell through to the full netmap rebuild path because they
// carried a field that the incremental path can't handle. See
// [netmap.mapResponseContainsNonPatchFields].
metricMapResponseHandledFullRebuild = clientmetric.NewCounter("controlclient_map_response_handled_full_rebuild")
)
// updatePeersStateFromResponseres updates ms.peers from resp.