From 8afcea0287dea28013336e61d58e4ee8e38f4ff1 Mon Sep 17 00:00:00 2001 From: Marvin von Rappard Date: Wed, 23 Sep 2026 23:52:23 +0200 Subject: [PATCH] refactor(cloud): stop sending the retired tailnet peer-liveness frame The cloud no longer reads local-netmap peer liveness: it accepts the `tailnet` frame only for compatibility and discards it undecoded. Stop reading peers from `tailscale status` and sending the frame every 30 s. The same loop also kept this node's MagicDNS name (the Funnel destination) fresh, so that part stays as selfDNSNameLoop on the heartbeat cadence. The proto types are unchanged, so the wire contract is untouched. Refs marvinvr/docktail-cloud#100 --- cloud/collector.go | 4 +-- cloud/tailnet.go | 67 ++++++++++++++------------------------------- tailscale/status.go | 45 ++++++------------------------ 3 files changed, 32 insertions(+), 84 deletions(-) diff --git a/cloud/collector.go b/cloud/collector.go index d55a718..201b1a8 100644 --- a/cloud/collector.go +++ b/cloud/collector.go @@ -100,7 +100,7 @@ type containerStats struct { // NewCollector builds a Collector, reading the host fingerprint (docker engine // ID) and versions up front. Returns an error only if the engine ID can't be // read — without it there is no stable host identity. ts reads the local -// tailscale daemon (peer liveness, node identity) and, when API credentials are +// tailscale daemon (node identity, MagicDNS name) and, when API credentials are // configured, the Tailscale control plane behind the tailnet vantage; pass nil // to run without any tailnet signals. func NewCollector(ctx context.Context, cfg Config, dc *docker.Client, ts tailnetSource, logger zerolog.Logger) (*Collector, error) { @@ -759,7 +759,7 @@ func (c *Collector) session(ctx context.Context, bo *backoff) (stop bool) { go c.metricsLoop(connCtx, conn) } if c.tailnet != nil { - go c.tailnetLoop(connCtx, conn) + go c.selfDNSNameLoop(connCtx) } err = <-runDone diff --git a/cloud/tailnet.go b/cloud/tailnet.go index 488a601..156a905 100644 --- a/cloud/tailnet.go +++ b/cloud/tailnet.go @@ -13,14 +13,14 @@ import ( ) // tailnetSource is the read-only tailscale view the collector needs: the local -// daemon (peer liveness, this node's identity, the tailnet name) plus the +// daemon (this node's identity, MagicDNS name and tailnet name) plus the // control-plane reads that answer a [proto.TailnetProbe] with the credentials // DockTail already holds. *tailscale.Client satisfies it. Nil when DockTail has -// no tailscale client, in which case the collector reports no peer liveness and -// answers every probe as unavailable. +// no tailscale client, in which case the collector reports no tailnet identity +// and answers every probe as unavailable. type tailnetSource interface { - // Status returns this node's stable ID, its tailnet name, and the liveness - // of the tailnet peers it can see, from `tailscale status`. + // Status returns this node's stable ID, MagicDNS name and tailnet name, + // from `tailscale status`. Status(ctx context.Context) (*tailscale.TailnetStatus, error) // APIEnabled reports whether Tailscale API credentials are configured. // False ⇒ the control plane is unreadable and no probe can be answered. @@ -254,10 +254,9 @@ func controlErrorText(err error) string { // tailnetIdentity reads this node's tailscale StableNodeID and tailnet name // (best-effort, bounded) for the hello frame, in ONE `tailscale status` call. -// The node ID is how the cloud splits THIS host's outages into agent_down vs -// host_down (empty ⇒ it falls back to host_down) and how it matches this host -// against a service's control-plane hosts; the tailnet name groups hosts that -// share a control plane, so the cloud can probe one of them on behalf of all. +// The node ID is how the cloud matches this host against a service's +// control-plane hosts; the tailnet name groups hosts that share a control +// plane, so the cloud can probe one of them on behalf of all. func (c *Collector) tailnetIdentity(ctx context.Context) (nodeID, tailnet string) { if c.tailnet == nil { return "", "" @@ -275,8 +274,8 @@ func (c *Collector) tailnetIdentity(ctx context.Context) (nodeID, tailnet string // funnelHostname is this node's MagicDNS name as last read from the local // daemon, or "" when there is none. It is read on the snapshot path, so it is // cached rather than shelled out per service: the name changes about as -// often as the machine is renamed, and the heartbeat-cadence netmap read already -// refreshes it. +// often as the machine is renamed, and [Collector.selfDNSNameLoop] refreshes it +// on the heartbeat cadence. func (c *Collector) funnelHostname() string { c.mu.RLock() defer c.mu.RUnlock() @@ -295,59 +294,35 @@ func (c *Collector) setSelfDNSName(name string) { c.mu.Unlock() } -// tailnetLoop reports the host's local-netmap peer liveness on the heartbeat -// cadence. It is the signal the cloud uses to tell a dead agent (the device is -// still online) from a dead host (device gone) for OTHER hosts on the tailnet. -// Skipped while unmonitored (the cloud drops the frame) and silent when there is -// no tailnet. -func (c *Collector) tailnetLoop(ctx context.Context, conn *wsConn) { +// selfDNSNameLoop re-reads this node's MagicDNS name on the heartbeat cadence, +// so a node that only just got one (a late login, MagicDNS switched on) starts +// reporting its Funnel destination without a reconnect. Skipped while +// unmonitored: the cloud runs no checks for such a host. +func (c *Collector) selfDNSNameLoop(ctx context.Context) { ticker := time.NewTicker(proto.HeartbeatInterval * time.Second) defer ticker.Stop() - c.sampleAndSendTailnet(ctx, conn) for { select { case <-ctx.Done(): return case <-ticker.C: - c.sampleAndSendTailnet(ctx, conn) + c.refreshSelfDNSName(ctx) } } } -func (c *Collector) sampleAndSendTailnet(ctx context.Context, conn *wsConn) { - if c.tailnet == nil { - return - } +func (c *Collector) refreshSelfDNSName(ctx context.Context) { c.mu.RLock() unmonitored := c.unmonitored c.mu.RUnlock() if unmonitored { return } - st, err := c.tailnet.Status(ctx) + cctx, cancel := context.WithTimeout(ctx, 5*time.Second) + defer cancel() + st, err := c.tailnet.Status(cctx) if err != nil || st == nil { - return // no tailnet → nothing to report + return // no tailnet → keep whatever name we last saw } - // Same read, second use: keep the funnel destination fresh on the heartbeat - // cadence so a node that only just got its MagicDNS name starts reporting one. c.setSelfDNSName(st.SelfDNSName) - peers := make([]proto.TailnetPeer, 0, len(st.Peers)) - for _, p := range st.Peers { - if p.NodeID == "" { - continue - } - tp := proto.TailnetPeer{ - NodeID: p.NodeID, - Hostname: p.Hostname, - Online: p.Online, - } - if !p.Online && !p.LastSeen.IsZero() { - tp.LastSeen = p.LastSeen.UnixMilli() - } - peers = append(peers, tp) - } - if len(peers) == 0 { - return - } - c.send(conn, proto.TypeTailnet, proto.TailnetReport{Peers: peers}) } diff --git a/tailscale/status.go b/tailscale/status.go index 76b24be..033fe3b 100644 --- a/tailscale/status.go +++ b/tailscale/status.go @@ -5,15 +5,12 @@ import ( "encoding/json" "fmt" "strings" - "time" ) // TailnetStatus is a minimal view of `tailscale status --json`: this node's -// stable ID and MagicDNS name, the tailnet it belongs to, and the online/offline -// status of the tailnet devices (peers) it can see. DockTail Cloud uses it (read -// over the local daemon, no API key) to report peer device liveness so the cloud -// can tell a dead agent from a dead host, and to say where this node's Funnel -// exposure answers on the public internet. +// stable ID and MagicDNS name, and the tailnet it belongs to. DockTail Cloud +// reads it over the local daemon (no API key) to identify this node and to say +// where its Funnel exposure answers on the public internet. type TailnetStatus struct { SelfNodeID string // SelfDNSName is this node's MagicDNS name with the trailing dot stripped @@ -25,23 +22,13 @@ type TailnetStatus struct { // daemons and a logged-out node report nothing, so an empty value means // "unknown", never "no tailnet". Tailnet string - Peers []TailnetPeerStatus -} - -// TailnetPeerStatus is one tailnet device's liveness as seen in this node's netmap. -type TailnetPeerStatus struct { - NodeID string - Hostname string - Online bool - LastSeen time.Time } // statusJSON is the subset of `tailscale status --json` (tailscaled's // ipnstate.Status) that we parse. type statusJSON struct { - Self *statusNode `json:"Self"` - Peer map[string]*statusNode `json:"Peer"` - CurrentTailnet *statusTailnet `json:"CurrentTailnet"` + Self *statusNode `json:"Self"` + CurrentTailnet *statusTailnet `json:"CurrentTailnet"` } // statusTailnet is ipnstate.Status.CurrentTailnet. Name is the human-facing @@ -53,15 +40,12 @@ type statusTailnet struct { } type statusNode struct { - ID string `json:"ID"` - DNSName string `json:"DNSName"` // FQDN with a trailing dot, e.g. "box.tail1234.ts.net." - HostName string `json:"HostName"` - Online bool `json:"Online"` - LastSeen time.Time `json:"LastSeen"` + ID string `json:"ID"` + DNSName string `json:"DNSName"` // FQDN with a trailing dot, e.g. "box.tail1234.ts.net." } -// Status runs `tailscale status --json` and parses this node's stable ID, its -// tailnet name, and the liveness of the peers it can see. Returns an error if +// Status runs `tailscale status --json` and parses this node's stable ID, +// MagicDNS name and tailnet name. Returns an error if // the daemon isn't reachable or the output can't be parsed — callers treat that // as "no tailnet" and skip reporting. func (c *Client) Status(ctx context.Context) (*TailnetStatus, error) { @@ -85,16 +69,5 @@ func (c *Client) Status(ctx context.Context) (*TailnetStatus, error) { out.Tailnet = st.CurrentTailnet.MagicDNSSuffix } } - for _, p := range st.Peer { - if p == nil || p.ID == "" { - continue - } - out.Peers = append(out.Peers, TailnetPeerStatus{ - NodeID: p.ID, - Hostname: p.HostName, - Online: p.Online, - LastSeen: p.LastSeen, - }) - } return out, nil }