tka/sync: add regression test for compacted nodes on forked chains

We previously identified sync failures that occur when a node falls behind
the remote, and compacts away most its local state. We fixed the underlying
issue in #19444, but that PR only tested the basic scenario where the
local chain is a direct ancestor of the remote chain.

This patch adds an explicit regression test for the case where a node is on
a fork (that is, its HEAD is not part of the remote's active chain).

Although #19444 happened to cover this case, other proposed patches did not
handle the forked state. Adding this test locks in the behaviour and prevents
future sync regressions in this area.

Also, add a shared helper for writing this sort of TKA sync test.

Updates tailscale/corp#40404

Change-Id: I78fdc6beaf71392edf11806197f126db48886f93
Signed-off-by: Alex Chan <alexc@tailscale.com>
This commit is contained in:
Alex Chan
2026-07-24 16:42:42 +01:00
committed by Alex Chan
parent 8c98d2a417
commit b5fb042501
2 changed files with 154 additions and 25 deletions
+58
View File
@@ -9,6 +9,8 @@ import (
"errors"
"fmt"
"os"
"tailscale.com/util/testenv"
)
// ErrNoIntersection is returned when a shared AUM could
@@ -257,3 +259,59 @@ func (a *Authority) MissingAUMs(storage Chonk, remoteOffer SyncOffer) ([]AUM, er
panic("unreachable")
}
// seedNode is an authority-chonk pair that can be seeded by [SeedAUMs].
type seedNode struct {
authority *Authority
storage Chonk
}
// CreateSeedNode creates a node for use with [SeedAUMs].
func CreateSeedNode(t testenv.TB, authority *Authority, storage Chonk) seedNode {
t.Helper()
return seedNode{authority, storage}
}
// SeedAUMs generates many AUMs by repeatedly adding and removing keys
// from the TKA.
//
// The AUMs are written to all the supplied nodes, so if you pass more
// than one, you can build up a long sync history.
//
// This is only for use in testing.
func SeedAUMs(t testenv.TB, count int, signer Signer, nodes ...seedNode) {
t.Helper()
if len(nodes) == 0 {
panic("called SeedAUMs without any nodes")
}
primaryNode := nodes[0]
// The key that we'll repeatedly add/remove in the TKA.
key := Key{Kind: Key25519, Public: []byte{1, 1, 1}, Votes: 1}
for i := 0; i < count/2; i++ {
for _, action := range []string{"add", "remove"} {
updater := primaryNode.authority.NewUpdater(signer)
if action == "add" {
if err := updater.AddKey(key); err != nil {
t.Fatalf("error from updater.AddKey: %v")
}
} else {
if err := updater.RemoveKey(key.MustID()); err != nil {
t.Fatalf("error from updater.RemoveKey: %v")
}
}
aum, err := updater.Finalize(primaryNode.storage)
if err != nil {
t.Fatalf("error from authority.Finalize: %v", err)
}
for _, n := range nodes {
if err := n.authority.Inform(n.storage, aum); err != nil {
t.Fatalf("error from authority.Inform: %v", err)
}
}
}
}
}
+96 -25
View File
@@ -435,11 +435,9 @@ func TestSyncSimpleE2E(t *testing.T) {
// Regression test for http://go/corp/40404
func TestSyncFromFarBehind(t *testing.T) {
pub1, priv1 := testingKey25519(t, 1)
pub2, _ := testingKey25519(t, 2)
signer1 := signer25519(priv1)
key1 := Key{Kind: Key25519, Public: pub1, Votes: 2}
key2 := Key{Kind: Key25519, Public: pub2, Votes: 2}
// Setup: persistentAuthority (control plane) vs compactingAuthority (client node).
state := State{
@@ -462,20 +460,9 @@ func TestSyncFromFarBehind(t *testing.T) {
compactingAuthority := must.Get(Bootstrap(compactingStorage, genesisAUM))
// 1. Generate enough history to trigger checkpoints.
for range checkpointEvery * 2 {
update := persistentAuthority.NewUpdater(signer1)
must.Do(update.AddKey(key2))
addKey := must.Get(update.Finalize(persistentStorage))
must.Do(persistentAuthority.Inform(persistentStorage, addKey))
must.Do(compactingAuthority.Inform(compactingStorage, addKey))
update = persistentAuthority.NewUpdater(signer1)
must.Do(update.RemoveKey(key2.MustID()))
removeKey := must.Get(update.Finalize(persistentStorage))
must.Do(persistentAuthority.Inform(persistentStorage, removeKey))
must.Do(compactingAuthority.Inform(compactingStorage, removeKey))
}
persistentNode := CreateSeedNode(t, persistentAuthority, persistentStorage)
compactingNode := CreateSeedNode(t, compactingAuthority, compactingStorage)
SeedAUMs(t, checkpointEvery*2, signer1, persistentNode, compactingNode)
t.Logf("genesis and first batch of AUMs: persistent = %d, compacting = %d", persistentSize(), compactingSize())
@@ -496,18 +483,102 @@ func TestSyncFromFarBehind(t *testing.T) {
//
// If you keep increasing this number, eventually the sync will fail because you
// hit the hard-coded limits on iteration during the sync process.
for persistentSize() < compactingSize()+800 {
b := persistentAuthority.NewUpdater(signer1)
SeedAUMs(t, compactingSize()-persistentSize()+800, signer1, persistentNode)
must.Do(b.AddKey(key2))
addKey := must.Get(b.Finalize(persistentStorage))
must.Do(persistentAuthority.Inform(persistentStorage, addKey))
t.Logf("post-compacting and extra AUMs: persistent = %d, compacting = %d", persistentSize(), compactingSize())
b = persistentAuthority.NewUpdater(signer1)
must.Do(b.RemoveKey(key2.MustID()))
removeKey := must.Get(b.Finalize(persistentStorage))
must.Do(persistentAuthority.Inform(persistentStorage, removeKey))
// 4. Verify Intersection.
// The node should find an intersection even with a 500-AUM gap.
persistentOffer := must.Get(persistentAuthority.SyncOffer(persistentStorage))
compactingOffer := must.Get(compactingAuthority.SyncOffer(compactingStorage))
if _, err := compactingAuthority.MissingAUMs(compactingStorage, persistentOffer); err != nil {
t.Errorf("node failed to find intersection with far-ahead control plane: %v", err)
}
// 5. Check that the persistent authority can find an intersection with the
// compacting authority, and has missing AUMs to send it.
missing, err := persistentAuthority.MissingAUMs(persistentStorage, compactingOffer)
if len(missing) == 0 {
t.Errorf("control plane did not find any missing AUMs for node")
}
if err != nil {
t.Errorf("control plane failed to find missing AUMs for node: %v", err)
}
}
// TestSyncFromFarBehindFork checks that nodes with compacted state that have
// also branched from the active chain can still find a common ancestor when
// the remote is significantly ahead of the inersection point.
//
// We simulate a node that has compacted its early history and is now ~500 AUMs
// behind the control plane, plus a few AUMs extra, a distance that previously
// caused exponential sampling in SyncOffer to skip the node's entire local history.
//
// Regression test for http://go/corp/40404
func TestSyncFromFarBehindFork(t *testing.T) {
// Set up two signing keys. They have a different number of votes, so if there's
// a fork, the chain with the winning key will take precedence.
majorityPub, majorityPriv := testingKey25519(t, 1)
losingPub, losingPriv := testingKey25519(t, 2)
winningSigner := signer25519(majorityPriv)
losingSigner := signer25519(losingPriv)
losingKey := Key{Kind: Key25519, Public: majorityPub, Votes: 1}
winningKey := Key{Kind: Key25519, Public: losingPub, Votes: 2}
// Setup: persistentAuthority (control plane) vs compactingAuthority (client node).
state := State{
Keys: []Key{losingKey, winningKey},
DisablementValues: [][]byte{DisablementKDF([]byte{1, 2, 3})},
}
persistentStorage, compactingStorage := ChonkMem(), ChonkMem()
persistentSize := func() int { return len(must.Get(persistentStorage.AllAUMs())) }
compactingSize := func() int { return len(must.Get(compactingStorage.AllAUMs())) }
// Backdate the clock on the compactingStorage so all AUMs will be old enough
// to be considered for compacting.
clock := tstest.NewClock(tstest.ClockOpts{
Start: time.Now().Add(-(CompactionDefaults.MinAge + 24*time.Hour)),
})
compactingStorage.SetClock(clock)
persistentAuthority, genesisAUM := must.Get2(Create(persistentStorage, state, winningSigner))
compactingAuthority := must.Get(Bootstrap(compactingStorage, genesisAUM))
// 1. Generate enough history to trigger checkpoints.
persistentNode := CreateSeedNode(t, persistentAuthority, persistentStorage)
compactingNode := CreateSeedNode(t, compactingAuthority, compactingStorage)
SeedAUMs(t, checkpointEvery*2, winningSigner, persistentNode, compactingNode)
t.Logf("genesis and first batch of AUMs: persistent = %d, compacting = %d", persistentSize(), compactingSize())
// 2. Compact the node state.
//
// It now has a different 'oldestAncestor' than the control plane.
beforeCompacting := compactingSize()
must.Do(compactingAuthority.Compact(compactingStorage, CompactionDefaults))
afterCompacting := compactingSize()
if beforeCompacting == afterCompacting {
t.Errorf("expected Compact to reduce the number of AUMs, but unchanged: size = %d", afterCompacting)
}
// 2. Advance the node state slightly beyond the control plane, using the
// losing signer.
SeedAUMs(t, 1, losingSigner, compactingNode)
// 3. Advance the control plane far beyond the node, using the winning signer.
//
// Now the node is forked from the control plane state, and the control plane's
// chain will win because its chain was signed by a key with more votes.
//
// As of 2026-04-17, the largest TKA has ~750 AUMs.
//
// If you keep increasing this number, eventually the sync will fail because you
// hit the hard-coded limits on iteration during the sync process.
SeedAUMs(t, compactingSize()-persistentSize()+800, winningSigner, persistentNode)
t.Logf("post-compacting and extra AUMs: persistent = %d, compacting = %d", persistentSize(), compactingSize())