From 4348edbd4170fbf39dda8bb16c585528d17034c7 Mon Sep 17 00:00:00 2001 From: Teo Calin Date: Tue, 6 Oct 2026 02:10:30 +0300 Subject: [PATCH 1/2] daemon: stop maintaining paths to peers nobody is using Every peer the daemon had exchanged keys with got a NAT keepalive every 25 s, path-watchdog probes and resets when quiet, and, when relayed, a direct-upgrade attempt every 15 s (beacon punch, five probes, a registry resolve each minute), forever. The keepalives also counted as contact, so the stale-peer reaper never fired. A service agent with ~4,000 peers and no open connection sent ~120 KB/s of this. Track application traffic per peer. A peer with none for 2 minutes and no open connection gets no keepalives, path probes or upgrade attempts; the reaper then drops it five minutes after its last frame. The reaper now counts relayed inbound frames as contact, so a peer still sending to us is kept. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 15 +++ pkg/daemon/daemon.go | 15 ++- pkg/daemon/pathwatch.go | 7 ++ pkg/daemon/peeractivity.go | 114 +++++++++++++++++ pkg/daemon/tunnel.go | 24 ++++ pkg/daemon/zz_idle_peer_upkeep_test.go | 167 +++++++++++++++++++++++++ 6 files changed, 341 insertions(+), 1 deletion(-) create mode 100644 pkg/daemon/peeractivity.go create mode 100644 pkg/daemon/zz_idle_peer_upkeep_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index fd723e61..aba8f82d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -245,6 +245,21 @@ Detailed per-release notes are on the unknown type; before it answered that no verifier was configured. ### Changed +- **Peers nobody is using no longer cost anything.** The daemon kept every + peer it had ever exchanged keys with warm forever: a NAT keepalive every + 25 s, path-watchdog probes and resets when the peer went quiet, and for + each relayed peer a direct-upgrade attempt (registry lookup, beacon punch, + five probes) every 15 s. A public service agent that had answered a few + thousand clients sent about 1,600 packets a second on this while serving + a few requests a minute. A peer with no application traffic for 2 minutes + and no open connection now gets none of it. Sending to it works as + before, re-establishing the path through the usual fallbacks if a NAT + mapping expired meanwhile. Five minutes after the last frame either way, + the stale-peer reaper now drops the peer, as it was meant to: the + keepalives counted as contact, so it never fired, and busy nodes held + every peer they had ever seen (service agents: 3,500 to 10,900). The next + contact runs a fresh key exchange. The reaper also counts relayed inbound + frames as contact now, so a peer still sending to us is kept. - **The tunnel socket asks the kernel for 4 MB buffers** in each direction instead of the default (about 200 KB on Linux). Every tunnel shares the one socket, and several streams sending at once overflowed it. The kernel caps diff --git a/pkg/daemon/daemon.go b/pkg/daemon/daemon.go index 0128113e..f11e9c65 100644 --- a/pkg/daemon/daemon.go +++ b/pkg/daemon/daemon.go @@ -667,6 +667,7 @@ func New(cfg Config) *Daemon { // peer trusted only through the registry now gets the redacted event, // which is the conservative side. d.tunnels.SetPeerTrustFn(d.handshakeTrusts) + d.tunnels.SetOpenConnPeers(d.ports.ActiveNodeIDs) d.ipc = NewIPCServer(cfg.SocketPath, d) // HandshakeService is wired post-construction by the composition // root via RegisterHandshakeService (T3.3 — handshake plugin moved @@ -5922,11 +5923,17 @@ func (d *Daemon) reapStalePeers() { lastDirect := d.tunnels.LastDirectRecv(p.NodeID) lastOutbound, ok := d.tunnels.LastOutboundSend(p.NodeID) - // Find the latest known contact. + // Find the latest known contact. Any authenticated inbound frame + // counts, relayed ones too (LastDirectRecv skips those): a peer + // still talking to us is not stale, and dropping it would only + // make its next frame trigger a rekey. latest := lastDirect if ok && lastOutbound.After(latest) { latest = lastOutbound } + if lastIn, ok := d.tunnels.LastInboundDecrypt(p.NodeID); ok && lastIn.After(latest) { + latest = lastIn + } if latest.IsZero() { // Never contacted — fresh peer entry, don't reap yet. @@ -6100,7 +6107,13 @@ func (d *Daemon) relayProbeLoop() { case <-d.stopCh: return case <-ticker.C: + idle := d.tunnels.idleFilter(time.Now()) for _, nodeID := range d.tunnels.RelayPeerIDs() { + if idle(nodeID) { + // An upgrade costs a registry lookup, a beacon punch + // and five probes; nobody is using this path. + continue + } go d.tryDirectUpgrade(nodeID) } } diff --git a/pkg/daemon/pathwatch.go b/pkg/daemon/pathwatch.go index 8aabe524..2d0920e1 100644 --- a/pkg/daemon/pathwatch.go +++ b/pkg/daemon/pathwatch.go @@ -124,8 +124,15 @@ func (d *Daemon) pathWatchTick(states map[uint32]*pathPeerState, now time.Time) defer recoverLayer("L4", "pathWatchTick", d.bus, nil) ready := d.tunnels.ReadyPeerIDs() + idle := d.tunnels.idleFilter(now) live := make(map[uint32]bool, len(ready)) for _, nodeID := range ready { + if idle(nodeID) { + // Nobody is using this peer: silence from it is expected, + // and probing or resetting it would only cost both sides + // packets and a key exchange. Its state is dropped below. + continue + } live[nodeID] = true st := states[nodeID] if st == nil { diff --git a/pkg/daemon/peeractivity.go b/pkg/daemon/peeractivity.go new file mode 100644 index 00000000..718c6ea3 --- /dev/null +++ b/pkg/daemon/peeractivity.go @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +package daemon + +import ( + "sync" + "sync/atomic" + "time" +) + +// PeerIdleAfter is how long a peer may go without application traffic +// before the daemon stops maintaining its path in the background. +// +// The daemon keeps a session warm with three periodic jobs: NAT keepalives +// every TunnelKeepaliveInterval, path-watchdog probes when the peer goes +// quiet, and a direct-path upgrade attempt (registry lookup, beacon punch, +// five probes) every RelayProbeInterval for each relayed peer. All three +// ran for every peer the daemon had ever exchanged keys with, forever: a +// public service agent that had answered a few thousand clients spent +// about 1,600 packets a second on them while serving a few requests a +// minute. +// +// Sending to an idle peer works as before: the first packet re-establishes +// the path through the usual fallbacks (address learning, blackhole +// detection, relay) if a NAT mapping expired in the meantime. Five minutes +// after the last frame either way, the stale-peer reaper drops the peer +// (reapStalePeers), which the keepalives used to prevent; the next contact +// then runs a fresh key exchange, as first contact does. A peer with an open +// connection is never idle. +var PeerIdleAfter = 2 * time.Minute + +// peerActivity records, per peer, when application traffic last went to or +// came from it. Path upkeep (keepalives, path probes, upgrade probes, key +// exchange) is not application traffic and does not count. +type peerActivity struct { + mu sync.RWMutex + m map[uint32]*atomic.Int64 // unix nanoseconds +} + +func newPeerActivity() *peerActivity { + return &peerActivity{m: make(map[uint32]*atomic.Int64)} +} + +func (a *peerActivity) note(nodeID uint32, now time.Time) { + a.mu.RLock() + v := a.m[nodeID] + a.mu.RUnlock() + if v == nil { + a.mu.Lock() + if v = a.m[nodeID]; v == nil { + v = new(atomic.Int64) + a.m[nodeID] = v + } + a.mu.Unlock() + } + v.Store(now.UnixNano()) +} + +// last reports the last activity, and false when none was ever recorded. +func (a *peerActivity) last(nodeID uint32) (time.Time, bool) { + a.mu.RLock() + v := a.m[nodeID] + a.mu.RUnlock() + if v == nil { + return time.Time{}, false + } + return time.Unix(0, v.Load()), true +} + +func (a *peerActivity) forget(nodeID uint32) { + a.mu.Lock() + delete(a.m, nodeID) + a.mu.Unlock() +} + +// noteAppActivity marks application traffic to or from a peer. +func (tm *TunnelManager) noteAppActivity(nodeID uint32) { + if tm.activity != nil { + tm.activity.note(nodeID, time.Now()) + } +} + +// SetOpenConnPeers installs the daemon's view of which peers have an open +// connection; such a peer is never idle however quiet the connection is. +func (tm *TunnelManager) SetOpenConnPeers(fn func() map[uint32]bool) { + tm.openConnPeers = fn +} + +// idleFilter returns a predicate reporting whether a peer is idle at now. +// It reads the open-connection set once, so callers sweeping many peers +// pay for it once per sweep. A peer with no recorded activity counts as +// active: every session records activity when it is first established, so +// only state that predates tracking falls in that case. +func (tm *TunnelManager) idleFilter(now time.Time) func(nodeID uint32) bool { + var open map[uint32]bool + if tm.openConnPeers != nil { + open = tm.openConnPeers() + } + return func(nodeID uint32) bool { + if PeerIdleAfter <= 0 || tm.activity == nil || open[nodeID] { + return false + } + last, ok := tm.activity.last(nodeID) + if !ok { + return false + } + return now.Sub(last) > PeerIdleAfter + } +} + +// PeerIdle reports whether one peer is idle now. +func (tm *TunnelManager) PeerIdle(nodeID uint32) bool { + return tm.idleFilter(time.Now())(nodeID) +} diff --git a/pkg/daemon/tunnel.go b/pkg/daemon/tunnel.go index 34aeb1e8..eb8004f3 100644 --- a/pkg/daemon/tunnel.go +++ b/pkg/daemon/tunnel.go @@ -85,6 +85,10 @@ type TunnelManager struct { sock transport.Transport // peers maps node_id → real UDP endpoint. Owned by L4 (routing). peers map[uint32]*net.UDPAddr + // activity and openConnPeers decide which peers are idle; see + // peeractivity.go. + activity *peerActivity + openConnPeers func() map[uint32]bool // envelope is the L5-owned per-peer crypto Store (named "envelope" // for historical reasons — pre-T5.x-followup the Store lived in // pkg/daemon/envelope). Accessed by: @@ -404,6 +408,7 @@ func NewTunnelManager() *TunnelManager { routing: routing.New(), kxRateLim: make(map[string]*srcKxBucket), relayKxLim: make(map[uint32]*srcKxBucket), + activity: newPeerActivity(), } tm.routing.SetLocalNodeIDFn(tm.loadNodeID) tm.kx = keyexchange.New(store) @@ -974,6 +979,7 @@ func (tm *TunnelManager) keepaliveSweep(now time.Time) int { addr *net.UDPAddr pc *peerCrypto } + idle := tm.idleFilter(now) tm.mu.RLock() stale := make([]peerInfo, 0, len(tm.peers)) for nodeID, addr := range tm.peers { @@ -981,6 +987,9 @@ func (tm *TunnelManager) keepaliveSweep(now time.Time) int { if ok && now.Sub(last) < TunnelKeepaliveInterval { continue } + if idle(nodeID) { + continue + } pc := tm.envelope.Get(nodeID) if pc == nil || !pc.Ready { continue @@ -1478,6 +1487,12 @@ func (tm *TunnelManager) onKeyInstalled(ev keyexchange.PostInstallEvent) { // and the per-peer 1s reply cooldown holds the ping-pong back but // never lets the staleness flag clear. tm.recordInboundDecrypt(peerNodeID) + if !ev.HadCrypto { + // A new session counts as activity, so it starts out maintained + // and ages into idleness like any other. A rekey of an existing + // session does not: it is upkeep, not use. + tm.noteAppActivity(peerNodeID) + } if !ev.HadCrypto || ev.KeyChanged { tm.keyViaRelay.Store(peerNodeID, fromRelay) } @@ -1759,6 +1774,9 @@ func (tm *TunnelManager) handleEncrypted(data []byte, from *net.UDPAddr) { if isTunnelKeepalive(pkt) { return } + if !(pkt.Protocol == protocol.ProtoControl && pkt.DstPort == protocol.PortPing) { + tm.noteAppActivity(peerNodeID) + } select { case tm.recvCh <- &IncomingPacket{Packet: pkt, From: from}: @@ -2079,6 +2097,9 @@ func (tm *TunnelManager) SetRelayPeerPinned(nodeID uint32, relay bool) { // SendTo sends a packet to a specific UDP address (relay-aware). func (tm *TunnelManager) SendTo(addr *net.UDPAddr, nodeID uint32, pkt *protocol.Packet) error { + // Path upkeep (keepalives, probes, key exchange) bypasses SendTo, so + // everything that arrives here is traffic someone asked for. + tm.noteAppActivity(nodeID) data, err := pkt.Marshal() if err != nil { return fmt.Errorf("marshal: %w", err) @@ -2216,6 +2237,9 @@ func (tm *TunnelManager) RemovePeer(nodeID uint32) { // L5-owned per-peer state (peerPubKeys, pendingRekey, lastInboundDecrypt). tm.kx.RemovePeer(nodeID) tm.keyViaRelay.Delete(nodeID) + if tm.activity != nil { + tm.activity.forget(nodeID) + } } // KeyArrivedViaRelayOnly reports whether the peer's session key was diff --git a/pkg/daemon/zz_idle_peer_upkeep_test.go b/pkg/daemon/zz_idle_peer_upkeep_test.go new file mode 100644 index 00000000..1c33a6fd --- /dev/null +++ b/pkg/daemon/zz_idle_peer_upkeep_test.go @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +package daemon + +import ( + "crypto/ecdh" + "crypto/rand" + "net" + "testing" + "time" + + "github.com/pilot-protocol/common/protocol" +) + +// readyPeer installs a peer with a ready session whose last outbound send +// was long ago, so keepaliveSweep would normally refresh it. +func readyPeer(t *testing.T, tm *TunnelManager, nodeID uint32) { + t.Helper() + sock, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.ParseIP("127.0.0.1"), Port: 0}) + if err != nil { + t.Fatalf("peer listen: %v", err) + } + t.Cleanup(func() { sock.Close() }) + priv, err := ecdh.X25519().GenerateKey(rand.Reader) + if err != nil { + t.Fatalf("keygen: %v", err) + } + pc, err := tm.deriveSecret(priv.PublicKey().Bytes()) + if err != nil { + t.Fatalf("deriveSecret: %v", err) + } + pc.Ready = true + tm.mu.Lock() + tm.peers[nodeID] = sock.LocalAddr().(*net.UDPAddr) + tm.envelope.Install(nodeID, pc) + tm.routing.RecordOutboundSend(nodeID, time.Now().Add(-time.Minute)) + tm.mu.Unlock() +} + +func newEncryptedTunnel(t *testing.T) *TunnelManager { + t.Helper() + tm := NewTunnelManager() + t.Cleanup(func() { tm.Close() }) + if err := tm.Listen("127.0.0.1:0"); err != nil { + t.Fatalf("Listen: %v", err) + } + if err := tm.EnableEncryption(); err != nil { + t.Fatalf("EnableEncryption: %v", err) + } + tm.SetNodeID(0xAA000002) + return tm +} + +// Keepalives go to peers in use and to peers with an open connection, and +// not to peers nobody has used for longer than PeerIdleAfter. +func TestKeepaliveSweepSkipsIdlePeers(t *testing.T) { + t.Parallel() + tm := newEncryptedTunnel(t) + + const active, idle, idleOpenConn, untracked = 0xB1, 0xB2, 0xB3, 0xB4 + for _, id := range []uint32{active, idle, idleOpenConn, untracked} { + readyPeer(t, tm, id) + } + now := time.Now() + tm.activity.note(active, now.Add(-10*time.Second)) + tm.activity.note(idle, now.Add(-PeerIdleAfter-time.Minute)) + tm.activity.note(idleOpenConn, now.Add(-PeerIdleAfter-time.Minute)) + tm.SetOpenConnPeers(func() map[uint32]bool { return map[uint32]bool{idleOpenConn: true} }) + + if sent := tm.keepaliveSweep(now); sent != 3 { + t.Fatalf("keepalives sent = %d, want 3 (active, open connection, untracked; not the idle peer)", sent) + } + if last, _ := tm.routing.LastOutboundSend(idle); now.Sub(last) < 30*time.Second { + t.Fatal("the idle peer was sent a keepalive") + } +} + +// Application traffic in either direction makes a peer active; path +// upkeep does not. +func TestOnlyApplicationTrafficCountsAsActivity(t *testing.T) { + t.Parallel() + tm := newEncryptedTunnel(t) + const peer = 0xC1 + readyPeer(t, tm, peer) + old := time.Now().Add(-PeerIdleAfter - time.Minute) + tm.activity.note(peer, old) + + // Upkeep: a keepalive and a path probe must not refresh activity. + tm.keepaliveSweep(time.Now()) + if err := tm.SendPathProbe(peer); err != nil { + t.Fatalf("SendPathProbe: %v", err) + } + if last, _ := tm.activity.last(peer); !last.Equal(time.Unix(0, old.UnixNano())) { + t.Fatal("keepalive or path probe counted as application activity") + } + if !tm.PeerIdle(peer) { + t.Fatal("peer should be idle") + } + + // Application traffic: any packet sent through Send. + _ = tm.Send(peer, &protocol.Packet{ + Version: protocol.Version, + Protocol: protocol.ProtoStream, + Src: protocol.Addr{Node: 0xAA000002}, + Dst: protocol.Addr{Node: peer}, + DstPort: protocol.PortDataExchange, + }) + if tm.PeerIdle(peer) { + t.Fatal("sending application data did not make the peer active") + } +} + +// The path watchdog leaves idle peers alone: no probes, no resets. +func TestPathWatchSkipsIdlePeers(t *testing.T) { + resets := swapPathResetForTest(t) + const peer = 61 + d := newPathWatchTestDaemon(t, peer) + // Long silent, and long unused. + d.tunnels.kx.RecordInboundDecrypt(peer) + states := map[uint32]*pathPeerState{} + silentNow := time.Now().Add(2 * pathSilenceThreshold) + d.tunnels.activity.note(peer, silentNow.Add(-PeerIdleAfter-time.Minute)) + + for i := 0; i < pathProbeMax+3; i++ { + d.pathWatchTick(states, silentNow.Add(time.Duration(i)*pathWatchTickInterval)) + } + if len(*resets) != 0 { + t.Fatalf("idle peer was reset %d times", len(*resets)) + } + if st := states[peer]; st != nil && st.probesSent != 0 { + t.Fatalf("idle peer was probed %d times", st.probesSent) + } + + // The same peer in use is still watched. + d.tunnels.activity.note(peer, silentNow) + for i := 0; i < pathProbeMax+2; i++ { + d.pathWatchTick(states, silentNow.Add(time.Duration(i)*pathWatchTickInterval)) + } + if len(*resets) == 0 { + t.Fatal("an active silent peer was not reset; the watchdog no longer works") + } +} + +// Once an idle peer gets no keepalives, the stale-peer reaper can drop it +// (keepalives used to count as contact and kept every peer forever). A +// peer still sending to us is kept, even when its frames come through the +// relay, which LastDirectRecv does not count. +func TestReaperDropsIdlePeersButKeepsTalkingOnes(t *testing.T) { + const silent, talking = 71, 72 + d := newPathWatchTestDaemon(t, silent) + d.ports = NewPortManager() + d.tunnels.AddPeer(talking, &net.UDPAddr{IP: net.IPv4(203, 0, 113, 8), Port: 4000}) + long := time.Now().Add(-10 * time.Minute) + for _, id := range []uint32{silent, talking} { + d.tunnels.routing.RecordOutboundSend(id, long) + d.tunnels.routing.RecordDirectRecv(id, long) + } + d.tunnels.kx.RecordInboundDecrypt(talking) // a relayed frame just now + + d.reapStalePeers() + if d.tunnels.HasPeer(silent) { + t.Fatal("a peer silent for 10 minutes was not reaped") + } + if !d.tunnels.HasPeer(talking) { + t.Fatal("a peer that just sent us a frame was reaped") + } +} From f1516f7644252c88374938a155cb4046cd55a10a Mon Sep 17 00:00:00 2001 From: Teo Calin Date: Tue, 6 Oct 2026 03:07:02 +0300 Subject: [PATCH 2/2] tests: idle peers behind a NAT, before and after the reap Two Docker scenarios on a cone NAT with a 10 s UDP conntrack timeout: idle 160 s (keepalives stopped, mapping expired) and idle 8 minutes (both sides reap each other), then an echo in each direction. Co-Authored-By: Claude Opus 5.5 --- .../local/test_nat_idle_peer_reaped.sh | 87 +++++++++++++++++++ .../local/test_nat_idle_peer_resume.sh | 76 ++++++++++++++++ 2 files changed, 163 insertions(+) create mode 100755 tests/integration/local/test_nat_idle_peer_reaped.sh create mode 100755 tests/integration/local/test_nat_idle_peer_resume.sh diff --git a/tests/integration/local/test_nat_idle_peer_reaped.sh b/tests/integration/local/test_nat_idle_peer_reaped.sh new file mode 100755 index 00000000..19eac250 --- /dev/null +++ b/tests/integration/local/test_nat_idle_peer_reaped.sh @@ -0,0 +1,87 @@ +#!/bin/bash +# A peer left unused past PeerIdleAfter (2 min) gets no keepalives, so five +# minutes later the stale-peer reaper forgets it (reapStalePeers: no open +# connection, no contact for 5 min). Before idle peers stopped getting +# keepalives the reaper never fired, as keepalives counted as contact. +# After the reap both sides have to build the path again, from scratch, +# and traffic must get through in both directions. +# +# Cone NAT with a 10 s UDP conntrack timeout, idle for 8 minutes. +source "$(dirname "$0")/nat_test_common.sh" +export NAT_MODE="conntrack_short" +cd "$(dirname "$0")" || exit 1 +trap cleanup_nat EXIT + +boot_nat_stack + +log_test "agents register" +REG=$(wait_registered 2 60 || echo "0") +if [ "${REG:-0}" -ge 2 ]; then + log_pass "$REG nodes" +else + log_fail "only $REG registered" + exit 1 +fi + +log_test "a->b initial echo, then mutual trust" +OUT=$(echo_rt agent-a agent-b "reap-probe-1") +if echo "$OUT" | grep -q "reap-probe-1"; then + log_pass "initial echo ok" +else + log_fail "initial echo failed" +fi +establish_trust agent-b agent-a >/dev/null + +NID_A=$(agent_node_id agent-a) +AGENT_A_ADDR=$(pilot_addr "$NID_A") + +log_test "b->a echo while in use" +OUT=$(echo_rt agent-b "$AGENT_A_ADDR" "reap-probe-2") +if echo "$OUT" | grep -q "reap-probe-2"; then + log_pass "echo ok" +else + log_fail "echo failed before idling" +fi + +log_test "idle 480 s (2 min to idle, 5 more to the reap)" +sleep 480 +log_pass "idle period elapsed" + +NID_B=$(agent_node_id agent-b) +has_peer() { $DC exec -T "$1" pilotctl --json peers 2>/dev/null | jq -e --argjson n "$2" '[.data.peers[]?.node_id] | index($n) != null' >/dev/null; } +log_test "the idle peers were forgotten" +GONE="" +has_peer agent-a "$NID_B" || GONE="$GONE a-forgot-b" +has_peer agent-b "$NID_A" || GONE="$GONE b-forgot-a" +echo " reaped:${GONE:- none}" +log_pass "peer tables after idle:${GONE:- both still listed}" + +log_test "b->a echo after idle: inbound to the NATed peer" +T0=$(date +%s) +OUT=$(echo_rt agent-b "$AGENT_A_ADDR" "reap-probe-3" 30s) +T1=$(date +%s) +if echo "$OUT" | grep -q "reap-probe-3"; then + log_pass "echo ok after idle in $((T1 - T0)) s" +else + log_fail "echo failed after idle ($((T1 - T0)) s): $(echo "$OUT" | head -c 200)" +fi + +log_test "a->b echo after idle: outbound from the NATed peer" +T0=$(date +%s) +OUT=$(echo_rt agent-a agent-b "reap-probe-4" 30s) +T1=$(date +%s) +if echo "$OUT" | grep -q "reap-probe-4"; then + log_pass "echo ok after idle in $((T1 - T0)) s" +else + log_fail "echo failed after idle ($((T1 - T0)) s): $(echo "$OUT" | head -c 200)" +fi + +log_test "no panics" +BAD=$($DC logs rendezvous agent-a agent-b nat-gw 2>&1 | grep -iE "panic|fatal|race detected" | head -3) +if [ -z "$BAD" ]; then + log_pass "clean logs" +else + log_fail "found: $BAD" +fi + +print_summary_and_exit diff --git a/tests/integration/local/test_nat_idle_peer_resume.sh b/tests/integration/local/test_nat_idle_peer_resume.sh new file mode 100755 index 00000000..03621764 --- /dev/null +++ b/tests/integration/local/test_nat_idle_peer_resume.sh @@ -0,0 +1,76 @@ +#!/bin/bash +# A peer left unused past PeerIdleAfter (2 min) gets no keepalives or path +# probes, so a NAT mapping toward it can expire. Traffic must still get +# through afterwards, in both directions, by the usual fallbacks. +# +# Cone NAT with a 10 s UDP conntrack timeout (as test_nat_conntrack_timeout), +# but idle for 160 s: past PeerIdleAfter, and far past the mapping's life. +source "$(dirname "$0")/nat_test_common.sh" +export NAT_MODE="conntrack_short" +cd "$(dirname "$0")" || exit 1 +trap cleanup_nat EXIT + +boot_nat_stack + +log_test "agents register" +REG=$(wait_registered 2 60 || echo "0") +if [ "${REG:-0}" -ge 2 ]; then + log_pass "$REG nodes" +else + log_fail "only $REG registered" + exit 1 +fi + +log_test "a->b initial echo, then mutual trust" +OUT=$(echo_rt agent-a agent-b "idle-probe-1") +if echo "$OUT" | grep -q "idle-probe-1"; then + log_pass "initial echo ok" +else + log_fail "initial echo failed" +fi +establish_trust agent-b agent-a >/dev/null + +NID_A=$(agent_node_id agent-a) +AGENT_A_ADDR=$(pilot_addr "$NID_A") + +log_test "b->a echo while in use" +OUT=$(echo_rt agent-b "$AGENT_A_ADDR" "idle-probe-2") +if echo "$OUT" | grep -q "idle-probe-2"; then + log_pass "echo ok" +else + log_fail "echo failed before idling" +fi + +log_test "idle 160 s (past the 2 min idle threshold and the 10 s conntrack timeout)" +sleep 160 +log_pass "idle period elapsed" + +log_test "b->a echo after idle: inbound to the NATed peer" +T0=$(date +%s) +OUT=$(echo_rt agent-b "$AGENT_A_ADDR" "idle-probe-3" 30s) +T1=$(date +%s) +if echo "$OUT" | grep -q "idle-probe-3"; then + log_pass "echo ok after idle in $((T1 - T0)) s" +else + log_fail "echo failed after idle ($((T1 - T0)) s): $(echo "$OUT" | head -c 200)" +fi + +log_test "a->b echo after idle: outbound from the NATed peer" +T0=$(date +%s) +OUT=$(echo_rt agent-a agent-b "idle-probe-4" 30s) +T1=$(date +%s) +if echo "$OUT" | grep -q "idle-probe-4"; then + log_pass "echo ok after idle in $((T1 - T0)) s" +else + log_fail "echo failed after idle ($((T1 - T0)) s): $(echo "$OUT" | head -c 200)" +fi + +log_test "no panics" +BAD=$($DC logs rendezvous agent-a agent-b nat-gw 2>&1 | grep -iE "panic|fatal|race detected" | head -3) +if [ -z "$BAD" ]; then + log_pass "clean logs" +else + log_fail "found: $BAD" +fi + +print_summary_and_exit