windows: adopt GP-managed NRPT catch-all

This commit is contained in:
Dev Scribe
2026-08-21 14:46:44 +07:00
committed by Cuong Manh Le
parent 779fe015f0
commit 4d026d836c
13 changed files with 3627 additions and 294 deletions
+74 -11
View File
@@ -172,6 +172,22 @@ type prog struct {
// On Windows: *wfpState, on macOS: *pfState, nil on other platforms.
dnsInterceptState any
// dnsInterceptMu serializes DNS intercept lifecycle transitions - start, stop and
// the health monitor's rebuild - and guards every write to dnsInterceptState, so a
// service stop can never interleave with a monitor-driven rebuild.
dnsInterceptMu sync.Mutex //lint:ignore U1000 used on windows
// dnsInterceptStopRequested is set while a stop waits for dnsInterceptMu. The
// health and recovery flows read it as a shutdown signal and abandon their work,
// rather than making the stop wait out their probe backoffs.
dnsInterceptStopRequested atomic.Bool //lint:ignore U1000 used on windows
// nrptTransitionMu makes one NRPT ownership transition - observe, mutate, signal,
// record owner - atomic against shutdown and against another transition. It is
// deliberately finer-grained than dnsInterceptMu: it is taken for the duration of a
// single transition, never across the recovery flows' probe backoffs.
nrptTransitionMu sync.Mutex //lint:ignore U1000 used on windows
// lastTunnelIfaces tracks the tunnel set included in the last successfully loaded
// pf anchor. Pending tunnel state is kept separately so failed PF work is retried
// instead of being mistaken for an applied update. Protected by mu.
@@ -220,15 +236,22 @@ type prog struct {
// existing delayed checks provide a trailing reconciliation after churn.
pfIgnoredChangeLastReconcile atomic.Int64 //lint:ignore U1000 used on darwin
// pfProbeExpected holds the domain name of a pending pf interception probe.
// When non-empty, the DNS handler checks incoming queries against this value
// and signals pfProbeCh if matched. The probe verifies that pf's rdr rules
// are actually translating packets (not just present in rule text).
pfProbeExpected atomic.Value // string
// pfProbeCh is signaled when the DNS handler receives the expected probe query.
// The channel is created by probePFIntercept() and closed when the probe arrives.
pfProbeCh atomic.Value // *chan struct{}
// interceptProbes maps the domain of each pending interception probe to the channel
// that probe waits on. A probe verifies that interception is actually translating or
// redirecting packets, not merely present in rule text: the DNS handler looks up
// incoming queries here and signals the matching waiter.
//
// It holds one entry per in-flight probe rather than a single slot, because probes do
// overlap - the health monitor, a handback and a heal cycle can each have one out at
// the same time - and a single slot means the last registration wins and the loser
// waits out its timeout for a query that was answered. A false failure then triggers
// recovery work that was not needed.
//
// Registrations are rare and lookups happen on every query, so the map is stored as
// an immutable snapshot behind an atomic: readers never take a lock, writers copy
// under interceptProbeMu.
interceptProbes atomic.Value // map[string]chan struct{}
interceptProbeMu sync.Mutex //lint:ignore U1000 written only by registerInterceptProbe, used on darwin/windows
// VPN DNS manager for split DNS routing when intercept mode is active.
vpnDNS *vpnDNSManager
@@ -424,7 +447,12 @@ func (p *prog) postRun() {
p.runningOnDomainController = isDC
mainLog.Load().Debug().Msgf("running on domain controller: %t, role: %d", p.runningOnDomainController, roleInt)
}
p.resetDNS(false, false)
// A Windows organization can install a GP-owned NRPT catch-all before
// starting ctrld. Detect that policy before resetDNS touches adapter DNS;
// startDNSIntercept will then prove the rule functionally before adopting it.
if !p.skipInitialDNSReset() {
p.resetDNS(false, false)
}
ns := ctrld.InitializeOsResolver(false)
mainLog.Load().Debug().Msgf("initialized OS resolver with nameservers: %v", ns)
p.setDNS()
@@ -907,7 +935,7 @@ func (p *prog) setDNS() {
mainLog.Load().Fatal().Msgf("invalid --intercept-mode value %q: must be 'off', 'dns', or 'hard'", interceptMode)
}
if interceptMode == "" || interceptMode == "off" {
interceptMode = cfg.Service.InterceptMode
interceptMode = p.configuredInterceptMode()
if interceptMode != "" && interceptMode != "off" {
mainLog.Load().Info().Msgf("Intercept mode enabled via config (intercept_mode = %q)", interceptMode)
}
@@ -927,6 +955,30 @@ func (p *prog) setDNS() {
// software that also manages DNS. See issue #489.
if dnsIntercept {
if err := startDNSInterceptFn(p); err != nil {
// This check comes first: it is the one failure where DNS already works
// without ctrld touching anything else, so neither the refusal below nor the
// fallback applies.
//
// An externally managed rule was proved - by probe, not by registry shape -
// to be routing DNS to this listener. Falling through would rewrite adapter
// DNS after explicitly preserving it, and DNS still works, so stop here.
//
// Only a verified route earns this. A rule that merely exists does not: if it
// is not actually routing and intercept failed too, the machine would be left
// with no NRPT, no WFP and no adapter fallback - that is, unfiltered - so
// every other failure takes the paths below.
if interceptFailedUnderExternalDNSPolicy(err) {
if interceptFailedWithVerifiedExternalDNS(err) {
mainLog.Load().Error().Err(err).Msg("DNS intercept mode failed but externally managed DNS policy is verified routing to ctrld — not falling back to interface DNS settings")
} else {
// Owned by external policy but not proved to route: DNS is not
// reaching ctrld. Adapter DNS still stays as the organization set it,
// and setDnsOK stays false, so this start reports as failed until a
// probe succeeds.
mainLog.Load().Error().Err(err).Msg("DNS intercept mode failed and externally managed DNS policy is not routing to ctrld — leaving interface DNS settings untouched; the service is not ready")
}
return
}
// Interface DNS cannot express a port: macOS interface settings and Windows
// NRPT rules both name a resolver by IP alone. So it is only a usable
// fallback when the listener actually bound :53. When something else owns
@@ -1037,6 +1089,17 @@ func (p *prog) setDNS() {
}
}
// configuredInterceptMode resolves the service's effective intercept mode without
// mutating package state. Platform startup preflights use the same precedence as
// setDNS so they do not make adapter-DNS decisions from a different mode value.
func (p *prog) configuredInterceptMode() string {
im := interceptMode
if im == "" || im == "off" {
im = p.cfg.Service.InterceptMode
}
return im
}
func (p *prog) setDnsForRunningIface(nameservers []string) (runningIface *net.Interface) {
if p.runningIface == "" {
return