state: send a node's DNS config with its own refresh

Drained at dispatch for nodeAttrs changes from any policy, user or node
update, and for hostname/OS changes feeding NextDNS device metadata.
This commit is contained in:
Kristoffer Dalby
2026-09-28 13:18:01 +00:00
committed by Kristoffer Dalby
parent 59030f380d
commit 357f34a778
6 changed files with 63 additions and 17 deletions
+2 -1
View File
@@ -19,6 +19,7 @@ import (
"os/signal"
"path/filepath"
"runtime"
"slices"
"strings"
"sync"
"syscall"
@@ -1150,7 +1151,7 @@ func readOrCreatePrivateKey(path string) (*key.MachinePrivate, error) {
// All change should be enqueued here and empty will be automatically
// ignored.
func (h *Headscale) Change(cs ...change.Change) {
h.mapBatcher.AddWork(cs...)
h.mapBatcher.AddWork(slices.Concat(cs, h.state.DrainSelfRefreshes())...)
}
// HTTPHandler returns an [http.Handler] for the [Headscale] control server.
+13 -4
View File
@@ -96,6 +96,10 @@ type PolicyManager struct {
// can tell the policy moved even when another goroutine (the
// NodeStore writer) applied the SetNodes.
nodesGen atomic.Uint64
// nodeAttrsPending mirrors len(nodeAttrsChanged) > 0 so the drain,
// called on every dispatched change, skips pm.mu when idle.
nodeAttrsPending atomic.Bool
}
// filterAndPolicy combines the compiled filter rules with policy content for hashing.
@@ -1934,6 +1938,10 @@ func (pm *PolicyManager) refreshNodeAttrsLocked(newMap map[types.NodeID]tailcfg.
pm.nodeAttrsMap = newMap
pm.nodeAttrsHashes = newHashes
pm.nodeAttrsChanged = append(pm.nodeAttrsChanged, changed...)
if len(pm.nodeAttrsChanged) > 0 {
pm.nodeAttrsPending.Store(true)
}
}
// NodeCapMap returns the policy-derived CapMap for the given node, or
@@ -1980,9 +1988,9 @@ func (pm *PolicyManager) NodeCapMaps() map[types.NodeID]tailcfg.NodeCapMap {
// NodesWithChangedCapMap returns the IDs of nodes whose nodeAttrs
// CapMap shifted across one or more [PolicyManager.updateLocked] calls
// since the last drain. The buffer drains on return. The mapper calls
// this once per [state.State.ReloadPolicy] to decide which nodes need
// a [change.SelfUpdate].
// since the last drain. The buffer drains on return.
// [state.State.DrainSelfRefreshes] calls this whenever changes are
// dispatched to decide which nodes need a [change.SelfUpdate].
//
// [PolicyManager.refreshNodeAttrsLocked] APPENDS to the buffer; the drain
// returns the union of every change since the previous read. A concurrent
@@ -1990,7 +1998,7 @@ func (pm *PolicyManager) NodeCapMaps() map[types.NodeID]tailcfg.NodeCapMap {
// [PolicyManager.SetPolicy] and a drain cannot silently lose the
// policy-reload diff.
func (pm *PolicyManager) NodesWithChangedCapMap() []types.NodeID {
if pm == nil {
if pm == nil || !pm.nodeAttrsPending.Load() {
return nil
}
@@ -1999,6 +2007,7 @@ func (pm *PolicyManager) NodesWithChangedCapMap() []types.NodeID {
out := pm.nodeAttrsChanged
pm.nodeAttrsChanged = nil
pm.nodeAttrsPending.Store(false)
return out
}
+4
View File
@@ -36,6 +36,10 @@ type mapRequestDelta struct {
// changed (see [peerHostinfo]). Only that forces a whole-node resend.
peerHostinfoChanged bool
// dnsMetadataChanged reports whether a Hostinfo field feeding the
// node's NextDNS device metadata (Hostname, OS) changed.
dnsMetadataChanged bool
// routesChanged reports whether announced routes (RoutableIPs)
// changed. Routes are policy and election inputs, so they are tracked
// on their own.
+40 -12
View File
@@ -191,6 +191,11 @@ type State struct {
// caller snapshot or resurrected by an update racing with deletion.
persistMu sync.Mutex
// selfRefresh holds nodes whose NextDNS device metadata changed, drained
// by [State.DrainSelfRefreshes].
selfRefresh []types.NodeID
selfRefreshMu sync.Mutex
// registerLocks serialises registration per machine key so concurrent
// registrations of the same machine resolve to a single node instead of
// racing the find-then-create section and each creating their own.
@@ -365,21 +370,11 @@ func (s *State) ReloadPolicy() ([]change.Change, error) {
// policies to not propagate correctly when switching between policy types.
s.nodeStore.RebuildPeerMaps()
// Nodes whose CapMap shifted get their self refresh from
// [State.DrainSelfRefreshes] when these changes are dispatched.
//nolint:prealloc // cs starts with one element and may grow
cs := []change.Change{change.PolicyChange()}
// Per-node selective self refresh for nodeAttrs. A broadcast
// [change.PolicyChange] re-renders peer lists and packet filters
// but never repopulates a node's own [tailcfg.Node.CapMap]; that
// lives on the self entry only. The drain returns every node ID
// whose cap output shifted across recent updateLocked calls —
// refreshNodeAttrsLocked appends rather than overwrites so a
// concurrent SetUsers/SetNodes between SetPolicy and the drain
// cannot silently lose the policy-reload diff.
for _, id := range s.polMan.NodesWithChangedCapMap() {
cs = append(cs, change.SelfUpdate(id))
}
// Always call autoApproveNodes during policy reload, regardless of whether
// the policy content has changed. This ensures that routes are re-evaluated
// when they might have been manually disabled but could now be auto-approved
@@ -3083,6 +3078,29 @@ func (s *State) UpdatePolicyManagerUsersForTest() error {
return err
}
// DrainSelfRefreshes returns a [change.SelfUpdate] for every node whose own
// entry or DNS config changed since the last call: its policy CapMap, from
// any policy, user or node update, or its NextDNS device metadata. Drained
// where changes are dispatched, so no path has to return the refresh itself.
func (s *State) DrainSelfRefreshes() []change.Change {
ids := s.polMan.NodesWithChangedCapMap()
s.selfRefreshMu.Lock()
ids = append(ids, s.selfRefresh...)
s.selfRefresh = nil
s.selfRefreshMu.Unlock()
slices.Sort(ids)
ids = slices.Compact(ids)
cs := make([]change.Change, 0, len(ids))
for _, id := range ids {
cs = append(cs, change.SelfUpdate(id))
}
return cs
}
// updatePolicyManagerNodes refreshes the policy manager with current node
// data and returns a PolicyChange when a node write since genBefore moved
// the policy. genBefore is [policy.PolicyManager.NodesGeneration] read
@@ -3332,6 +3350,10 @@ func (s *State) UpdateNodeFromMapRequest(id types.NodeID, req tailcfg.MapRequest
!hostinfoEqual(currentNode.Hostinfo, newHostinfo)
delta.peerHostinfoChanged = newHostinfo != nil &&
!peerHostinfoEqual(currentNode.Hostinfo, newHostinfo)
delta.dnsMetadataChanged = newHostinfo != nil &&
(currentNode.Hostinfo == nil ||
currentNode.Hostinfo.Hostname != newHostinfo.Hostname ||
currentNode.Hostinfo.OS != newHostinfo.OS)
// A change carrying only an updated LastSeen is not worth a
// full-row database UPDATE plus the O(n) policy rescan
@@ -3435,6 +3457,12 @@ func (s *State) UpdateNodeFromMapRequest(id types.NodeID, req tailcfg.MapRequest
return change.Change{}, fmt.Errorf("%w: %d", ErrNodeNotInNodeStore, id)
}
if delta.dnsMetadataChanged {
s.selfRefreshMu.Lock()
s.selfRefresh = append(s.selfRefresh, id)
s.selfRefreshMu.Unlock()
}
// SubnetRoutes = announced ∩ approved, so a Hostinfo update can
// move a primary without ever touching ApprovedRoutes. The pre/post
// snapshot diff catches that.
+3
View File
@@ -331,11 +331,14 @@ func FullSelf(nodeID types.NodeID) Change {
}
}
// SelfUpdate sends a node its own entry and the DNS config derived from it
// (NextDNS profile and device metadata).
func SelfUpdate(nodeID types.NodeID) Change {
return Change{
Reason: "self update",
TargetNode: nodeID,
IncludeSelf: true,
IncludeDNS: true,
}
}
+1
View File
@@ -464,6 +464,7 @@ func TestSelfUpdate(t *testing.T) {
assert.Equal(t, "self update", r.Reason)
assert.Equal(t, types.NodeID(42), r.TargetNode)
assert.True(t, r.IncludeSelf)
assert.True(t, r.IncludeDNS, "DNS config derives from the node's own CapMap and Hostinfo")
assert.True(t, r.IsSelfOnly())
}