From 7ec9275ca653f4179be04978ef366fada08aeac8 Mon Sep 17 00:00:00 2001 From: sophronesis Date: Mon, 28 Sep 2026 14:48:24 +0200 Subject: [PATCH] tailcat: mark magicsock's network up before the first netcheck TestServeExec and TestServeExitNode flaked in the Nix build sandbox (the x86_64-linux review of the v0.7.0 bump in NixOS/nixpkgs#565145) with "tailcat Ping: context deadline exceeded". With no non-loopback interface, wgengine starts magicsock with the network down. locoBackend.Start calls SetPrivateKey and SetDERPMap, which start endpoint updates in the background, and only later calls SetNetworkUp(true). If an endpoint update runs in between, updateNetInfo returns an empty report because the network is down, before it gets to maybeSetNearestDERP, so the server never picks a home DERP. SetNetworkUp(true) only connects to an already chosen home, so nothing fixes that until the periodic re-STUN 20-26s later, and a client's 10s Ping gives up first; its relay reports the server as not connected. Mark the network up first. When the host has a usable interface the network is already up and the call is a no-op. Reproduced by running the cmd/tailcat tests in a loopback-only network namespace (unshare -rn) with a 300ms sleep after SetDERPMap to widen the window: TestServeExec and TestServeExitNode time out without this change and pass with it. Without the sleep, the full cmd/tailcat suite passed 8 of 8 runs in that namespace with this change. Signed-off-by: sophronesis --- tailcat.go | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/tailcat.go b/tailcat.go index d110ffcf1..cc3c098dd 100644 --- a/tailcat.go +++ b/tailcat.go @@ -1628,6 +1628,14 @@ func (lb *locoBackend) Start() error { mc := lb.sys.MagicSock.Get() lb.logf("disco pub key: %v", mc.DiscoPublicKey()) + // Mark the network up before SetPrivateKey and SetDERPMap, which + // start endpoint updates in the background. With no non-loopback + // interface (as in the Nix build sandbox) magicsock starts with + // the network down, and an endpoint update that runs while it is + // still down skips netcheck and never picks a home DERP. A server + // then isn't reachable through DERP until the periodic re-STUN + // 20-26s later, long after clients give up. + mc.SetNetworkUp(true) mc.SetPrivateKey(lb.priv) mc.SetDERPMap(lb.dm) @@ -1683,7 +1691,6 @@ func (lb *locoBackend) Start() error { mc.SetNetworkMap(nm.SelfNode, nm.Peers) e.SetSelfNode(nm.SelfNode) lb.sys.Netstack.Get().UpdateNetstackIPs(nm) - mc.SetNetworkUp(true) lb.logf("NetworkMap: %v", logger.AsJSON(nm)) // Install the live per-peer config sources. WireGuard peers are