feature/captiveportal: move captive portal code out of ipnlocal, netcheck

Captive portal detection was half-migrated: it had a build tag and
buildfeatures constant, but its code still lived in build-tag-gated
files in ipn/ipnlocal and net/netcheck, with its per-backend state
(context, cancel func, signaling channel) as fields on LocalBackend.

Move it under feature/captiveportal. The health-driven detection loop
becomes an ipnext.Extension holding its own state: it starts and
stops the loop from the BackendStateChange hook and subscribes to
health.Change events on the eventbus itself, removing the captive
portal hooks and special cases from LocalBackend entirely. The DERP
map now comes from a new ipnext.NodeBackend.DERPMap method, and the
preferred DERP region from magicsock's last netcheck report (the
same underlying source as the previously used Hostinfo.NetInfo).

The netcheck probe hook is now exported with a signature free of
netcheck internals, and its implementation moves to the small
feature/captiveportal/netcheckhook package, which installs the hook
as an import side effect. That package stays free of tsd/wgengine
dependencies so the tailscale CLI can keep probing for captive
portals in "tailscale netcheck" without linking the daemon-side
extension. The net/captivedetection library itself is unchanged and
stays put; after this change it is only linked when something pulls
in netcheckhook or the feature extension.

tailscaled links the feature by default via condregister as before,
but tsnet no longer does (shrinking tsnet, k8s-operator, and tsidp);
tsnet users who want it can blank-import the feature package, and
tsnet's dep test now locks that in.

Updates #12614

Signed-off-by: Brad Fitzpatrick <bradfitz@tailscale.com>
Change-Id: I3f1d09f9dc03e18f9a648ab5e42d16fa540b3fa9
This commit is contained in:
Brad Fitzpatrick
2026-07-14 20:22:23 -04:00
committed by Brad Fitzpatrick
parent 72ca0cae4b
commit bb4f458207
15 changed files with 414 additions and 337 deletions
+285
View File
@@ -0,0 +1,285 @@
// Copyright (c) Tailscale Inc & contributors
// SPDX-License-Identifier: BSD-3-Clause
// Package captiveportal provides optional captive portal detection,
// warning the user via the health tracker when their network requires
// a browser login before traffic can flow.
package captiveportal
import (
"context"
"time"
"tailscale.com/feature"
_ "tailscale.com/feature/captiveportal/netcheckhook" // install the netcheck probe hook too
"tailscale.com/health"
"tailscale.com/ipn"
"tailscale.com/ipn/ipnext"
"tailscale.com/net/captivedetection"
"tailscale.com/syncs"
"tailscale.com/types/logger"
"tailscale.com/util/clientmetric"
"tailscale.com/util/eventbus"
)
const featureName = "captiveportal"
func init() {
feature.Register(featureName)
ipnext.RegisterExtension(featureName, newExtension)
}
var metricCaptivePortalDetected = clientmetric.NewCounter("captiveportal_detected")
// captivePortalDetectionInterval is the duration to wait in an unhealthy state with connectivity broken
// before running captive portal detection.
const captivePortalDetectionInterval = 2 * time.Second
// captivePortalWarnable is a Warnable which is set to an unhealthy state when a captive portal is detected.
var captivePortalWarnable = health.Register(&health.Warnable{
Code: "captive-portal-detected",
Title: "Captive portal detected",
// High severity, because captive portals block all traffic and require user intervention.
Severity: health.SeverityHigh,
Text: health.StaticMessage("This network requires you to log in using your web browser."),
ImpactsConnectivity: true,
})
// Extension is the captive portal detection extension.
// There is one per [ipnext.Host] (and hence per LocalBackend).
type Extension struct {
logf logger.Logf
sb ipnext.SafeBackend
host ipnext.Host // from Init
health *health.Tracker
ec *eventbus.Client
mu syncs.Mutex
// captiveCtx and captiveCancel are used to control captive portal
// detection. They are protected by 'mu' and can be changed during the
// lifetime of the extension.
//
// captiveCtx will always be non-nil, though it might be a canceled
// context. captiveCancel is non-nil if checkCaptivePortalLoop is
// running, and is set to nil after being canceled.
captiveCtx context.Context
captiveCancel context.CancelFunc
// needsCaptiveDetection is a channel that is used to signal either
// that captive portal detection is required (sending true) or that the
// backend is healthy and captive portal detection is not required
// (sending false).
needsCaptiveDetection chan bool
}
func newExtension(logf logger.Logf, sb ipnext.SafeBackend) (ipnext.Extension, error) {
// Initialize the context used to control the captive portal detection
// goroutine in a canceled state, so that it is always non-nil and safe
// to wait on even before the loop has ever started.
ctx, cancel := context.WithCancel(context.Background())
cancel()
return &Extension{
logf: logf,
sb: sb,
health: sb.Sys().HealthTracker.Get(),
captiveCtx: ctx,
needsCaptiveDetection: make(chan bool),
}, nil
}
func (e *Extension) Name() string { return featureName }
func (e *Extension) Init(h ipnext.Host) error {
e.host = h
h.Hooks().BackendStateChange.Add(e.onBackendStateChange)
e.ec = e.sb.Sys().Bus.Get().Client("captiveportal")
eventbus.SubscribeFunc(e.ec, e.onHealthChange)
return nil
}
func (e *Extension) Shutdown() error {
e.mu.Lock()
if e.captiveCancel != nil {
e.captiveCancel()
e.captiveCancel = nil
}
e.mu.Unlock()
if e.ec != nil {
e.ec.Close()
}
return nil
}
// onBackendStateChange starts or stops the captive portal detection loop as
// the backend enters or leaves the Running state. It is called with
// LocalBackend's mutex held, so it must not call back into LocalBackend.
func (e *Extension) onBackendStateChange(st ipn.State) {
e.mu.Lock()
defer e.mu.Unlock()
if st == ipn.Running {
// Start a captive portal detection loop if none has been
// started.
if e.captiveCancel == nil {
e.captiveCtx, e.captiveCancel = context.WithCancel(context.Background())
go e.checkCaptivePortalLoop(e.captiveCtx)
}
} else if e.captiveCancel != nil {
// Transitioning away from Running; stop any existing captive
// portal detection loop.
e.captiveCancel()
e.captiveCancel = nil
// NOTE: don't set captiveCtx to nil here, to ensure that we
// always have a (canceled) context to wait on in
// onHealthChange.
}
}
// onHealthChange is called (via the eventbus) whenever the node's health
// state changes. If connectivity appears to be impacted, it signals the
// detection loop to check for a captive portal.
func (e *Extension) onHealthChange(health.Change) {
state := e.health.CurrentState()
isConnectivityImpacted := false
for _, w := range state.Warnings {
// Ignore the captive portal warnable itself.
if w.ImpactsConnectivity && w.WarnableCode != captivePortalWarnable.Code {
isConnectivityImpacted = true
break
}
}
// captiveCtx can be changed, and is protected with 'mu'; grab that
// before we start our select, below.
//
// It is guaranteed to be non-nil.
e.mu.Lock()
ctx := e.captiveCtx
e.mu.Unlock()
// If the context is canceled, we don't need to do anything.
if ctx.Err() != nil {
return
}
if isConnectivityImpacted {
e.logf("health: connectivity impacted; triggering captive portal detection")
// Ensure that we select on captiveCtx so that we can time out
// triggering captive portal detection if the loop is shut down.
select {
case e.needsCaptiveDetection <- true:
case <-ctx.Done():
}
} else {
// If connectivity is not impacted, we know for sure we're not behind a captive portal,
// so drop any warning, and signal that we don't need captive portal detection.
e.health.SetHealthy(captivePortalWarnable)
select {
case e.needsCaptiveDetection <- false:
case <-ctx.Done():
}
}
}
func (e *Extension) checkCaptivePortalLoop(ctx context.Context) {
var tmr *time.Timer
maybeStartTimer := func() {
// If there's an existing timer, nothing to do; just continue
// waiting for it to expire. Otherwise, create a new timer.
if tmr == nil {
tmr = time.NewTimer(captivePortalDetectionInterval)
}
}
maybeStopTimer := func() {
if tmr == nil {
return
}
if !tmr.Stop() {
<-tmr.C
}
tmr = nil
}
for {
if ctx.Err() != nil {
maybeStopTimer()
return
}
// First, see if we have a signal on our "healthy" channel, which
// takes priority over an existing timer. Because a select is
// nondeterministic, we explicitly check this channel before
// entering the main select below, so that we're guaranteed to
// stop the timer before starting captive portal detection.
select {
case needsCaptiveDetection := <-e.needsCaptiveDetection:
if needsCaptiveDetection {
maybeStartTimer()
} else {
maybeStopTimer()
}
default:
}
var timerChan <-chan time.Time
if tmr != nil {
timerChan = tmr.C
}
select {
case <-ctx.Done():
// All done; stop the timer and then exit.
maybeStopTimer()
return
case <-timerChan:
// Kick off captive portal check
e.performCaptiveDetection(ctx)
// nil out the timer and its channel to force recreation
tmr, timerChan = nil, nil
case needsCaptiveDetection := <-e.needsCaptiveDetection:
if needsCaptiveDetection {
maybeStartTimer()
} else {
// Healthy; cancel any existing timer
maybeStopTimer()
}
}
}
}
// shouldRunCaptivePortalDetection reports whether captive portal detection
// should be run. It is enabled by default, but can be disabled via a control
// knob. It is also only run when the user explicitly wants the backend to be
// running.
func (e *Extension) shouldRunCaptivePortalDetection() bool {
return !e.sb.Sys().ControlKnobs().DisableCaptivePortalDetection.Load() &&
e.host.Profiles().CurrentPrefs().WantRunning()
}
// performCaptiveDetection checks if captive portal detection is enabled via controlknob. If so, it runs
// the detection and updates the Warnable accordingly.
func (e *Extension) performCaptiveDetection(ctx context.Context) {
if !e.shouldRunCaptivePortalDetection() {
return
}
d := captivedetection.NewDetector(e.logf)
dm := e.host.NodeBackend().DERPMap()
preferredDERP := 0
if mc, ok := e.sb.Sys().MagicSock.GetOK(); ok {
if report := mc.GetLastNetcheckReport(ctx); report != nil {
preferredDERP = report.PreferredDERP
}
}
netMon := e.sb.Sys().NetMon.Get()
found := d.Detect(ctx, netMon, dm, preferredDERP)
if found {
if !e.health.IsUnhealthy(captivePortalWarnable) {
metricCaptivePortalDetected.Add(1)
}
e.health.SetUnhealthy(captivePortalWarnable, health.Args{})
} else {
e.health.SetHealthy(captivePortalWarnable)
}
}
@@ -0,0 +1,67 @@
// Copyright (c) Tailscale Inc & contributors
// SPDX-License-Identifier: BSD-3-Clause
// Package netcheckhook makes netcheck probe for captive portals during
// full reports. It does so as a side effect of being imported, by
// installing a netcheck hook from init.
package netcheckhook
import (
"context"
"log"
"time"
"tailscale.com/net/captivedetection"
"tailscale.com/net/netcheck"
"tailscale.com/tailcfg"
)
func init() {
netcheck.HookStartCaptivePortalDetection.Set(startCaptivePortalDetection)
}
// captivePortalDelay is the duration to wait after starting a netcheck before
// also probing for a captive portal, to let UDP STUN finish first and avoid
// the probe if it's unnecessary. Chosen semi-arbitrarily.
const captivePortalDelay = 200 * time.Millisecond
func startCaptivePortalDetection(ctx context.Context, c *netcheck.Client, dm *tailcfg.DERPMap, preferredDERP int, setCaptivePortal func(bool)) (done <-chan struct{}, stop func()) {
logf := c.Logf
if logf == nil {
logf = log.Printf
}
// This goroutine can't be tracked by the wait group that
// netcheck.GetReport uses for its probes, since GetReport doesn't
// wait for that group to finish before returning and we'd get a
// data race. Instead, completion is signaled by closing ch, which
// GetReport receives as the done channel.
ch := make(chan struct{})
tmr := time.AfterFunc(captivePortalDelay, func() {
defer close(ch)
d := captivedetection.NewDetector(logf)
found := d.Detect(ctx, c.NetMon, dm, preferredDERP)
setCaptivePortal(found)
})
if c.Verbose {
// Don't cancel our captive portal check if we're
// explicitly doing a verbose netcheck.
return ch, func() {}
}
stop = func() {
if tmr.Stop() {
// Stopped successfully; need to close the
// signal channel ourselves.
close(ch)
return
}
// Did not stop; do nothing and it'll finish by itself
// and close the signal channel.
}
return ch, stop
}
@@ -0,0 +1,8 @@
// Copyright (c) Tailscale Inc & contributors
// SPDX-License-Identifier: BSD-3-Clause
//go:build !ts_omit_captiveportal
package condregister
import _ "tailscale.com/feature/captiveportal"