From 8fcfa88fe0fdb6c144e63a08cecda0b5415a0f91 Mon Sep 17 00:00:00 2001 From: jochen Date: Mon, 31 Aug 2026 10:21:26 +0200 Subject: [PATCH] A machine that wakes or moves says so, instead of waiting to be told MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A suspended laptop's connection is dead the moment it wakes, and the socket looks perfectly healthy from inside the process — no error, no close, because nothing has tried to send anything. Heartbeats find out twenty or thirty seconds later. For that time the node believes it is in a mesh it has left, which is the one state this design says must never be indistinguishable from being connected. The machine knew immediately. So being roused ends the current attempt rather than only shortening the wait after it: shortening the wait would do nothing at all, because the process is not waiting — it is sitting inside a connection that will not return. A signal, because nothing may listen on a node (novox/hq ADR 0004). A socket for this would be a control surface on every machine, reachable by anything that can reach the machine, in exchange for saving twenty seconds — and the whole security argument rests on there not being one. Two rouses in the same instant are one: a machine suspending and resuming repeatedly must not build a backlog of reconnections to work through. And the backoff is not reset by being roused — that says the machine changed, not that whatever was refusing the connection has stopped, and a laptop woken on a network with no route would otherwise retry at full speed for as long as somebody keeps opening the lid. The dispatcher acts on the events that change where packets go and not on `down`: the link is already gone there, reconnecting will fail, and the backoff exists for exactly that. --- Makefile | 1 + cmd/mesh-host/main.go | 40 +++++++++- internal/link/roused_test.go | 103 +++++++++++++++++++++++++ internal/link/run.go | 66 +++++++++++++++- packaging/nox-mesh-host-network.sh | 22 ++++++ packaging/nox-mesh-host-resume.service | 14 ++++ packaging/nox-mesh-host-roused.service | 17 ++++ packaging/roused_test.sh | 42 ++++++++++ 8 files changed, 302 insertions(+), 3 deletions(-) create mode 100644 internal/link/roused_test.go create mode 100755 packaging/nox-mesh-host-network.sh create mode 100644 packaging/nox-mesh-host-resume.service create mode 100644 packaging/nox-mesh-host-roused.service create mode 100755 packaging/roused_test.sh diff --git a/Makefile b/Makefile index e865709..ebe96e7 100644 --- a/Makefile +++ b/Makefile @@ -23,6 +23,7 @@ hosts: packaging-test: @./packaging/rollback_test.sh @./packaging/launch_test.sh + @./packaging/roused_test.sh fmt: @test -z "$$(gofmt -l . )" || { echo "unformatted:"; gofmt -l . ; exit 1; } diff --git a/cmd/mesh-host/main.go b/cmd/mesh-host/main.go index 2eb9eac..6fc4d63 100644 --- a/cmd/mesh-host/main.go +++ b/cmd/mesh-host/main.go @@ -602,13 +602,49 @@ func runLink(ctx context.Context, opts options) error { // (novox/hq ADR 0004). go holdTheMachine(ctx, opts, mine, say) - return link.Hold(ctx, link.Membership{ + return link.HoldRoused(ctx, link.Membership{ Node: mine.Node, Broker: mine.Membership.Broker, Fingerprint: mine.Membership.Fingerprint, Password: mine.Membership.Password, Signer: mine.Membership.Signer, - }, apply, say, opts.timeout) + }, apply, say, opts.timeout, rousedBySignal(ctx)) +} + +// rousedBySignal is the machine telling this process that its link is probably stale. +// +// **A signal, because nothing may listen on a node** (novox/hq ADR 0004). A socket for this would +// be a control surface on every machine, reachable by anything that can reach the machine, in +// exchange for saving twenty seconds — and the whole security argument rests on there not being +// one. A signal is delivered by the service manager to a process it already supervises. +// +// SIGHUP, because that is the signal a long-running program conventionally reads as *look again*, +// and nothing here is being reloaded from a file that a different signal would suit better. +// +// Dropped rather than queued when one arrives while another is unread: two wakes in the same +// instant are one wake, and a machine that suspends and resumes repeatedly must not build a +// backlog of reconnections to work through. +func rousedBySignal(ctx context.Context) link.Roused { + woken := make(chan os.Signal, 1) + signal.Notify(woken, syscall.SIGHUP) + + out := make(chan struct{}, 1) + go func() { + defer signal.Stop(woken) + for { + select { + case <-ctx.Done(): + return + case <-woken: + select { + case out <- struct{}{}: + default: + // One is already waiting to be read. Two wakes in the same instant are one. + } + } + } + }() + return out } // ReconcileEvery is how often a node re-applies what it was last told. diff --git a/internal/link/roused_test.go b/internal/link/roused_test.go new file mode 100644 index 0000000..1071566 --- /dev/null +++ b/internal/link/roused_test.go @@ -0,0 +1,103 @@ +package link + +import ( + "context" + "errors" + "strings" + "sync" + "testing" + "time" +) + +// A machine that just woke does not wait to be told its link is dead. +// +// After a resume the socket looks perfectly healthy from inside the process — no error, no close, +// because nothing has tried to send anything. Heartbeats discover it twenty or thirty seconds +// later, and for that time the node believes it is in a mesh it has left, which is the one state +// this design says must never be indistinguishable from being connected. +func TestBeingRousedEndsTheCurrentAttemptRatherThanWaitingForATimeout(t *testing.T) { + // A link that never returns on its own, which is exactly what a suspended connection is. + held := make(chan context.Context, 4) + running := func(ctx context.Context) error { + held <- ctx + <-ctx.Done() + return errors.New("the link ended") + } + + rouse := make(chan struct{}, 1) + said := &saidSoFar{} + ctx, stop := context.WithCancel(context.Background()) + defer stop() + + finished := make(chan error, 1) + go func() { finished <- holdWith(ctx, running, said.say, rouse) }() + + first := <-held + select { + case <-first.Done(): + t.Fatal("the link ended before anything roused it") + case <-time.After(50 * time.Millisecond): + } + + rouse <- struct{}{} + select { + case <-first.Done(): + case <-time.After(2 * time.Second): + t.Fatal("the machine woke and the link was left running against a socket that is gone") + } + + // And it opens another one rather than stopping. + select { + case <-held: + case <-time.After(5 * time.Second): + t.Fatal("the link was dropped and never opened again") + } + if !strings.Contains(said.all(), "woken or moved") { + t.Fatalf("nothing was said about why the link was dropped:\n%s", said.all()) + } + + stop() + select { + case <-finished: + case <-time.After(2 * time.Second): + t.Fatal("it did not stop when asked") + } +} + +// A machine nothing ever rouses is every machine that does not suspend, and must behave as before. +func TestAMachineNothingRousesIsUnaffected(t *testing.T) { + attempts := make(chan struct{}, 4) + running := func(context.Context) error { + attempts <- struct{}{} + return errors.New("the link ended") + } + ctx, stop := context.WithCancel(context.Background()) + defer stop() + go func() { _ = holdWith(ctx, running, func(string) {}, nil) }() + + // It keeps trying, which is the behaviour a node with no rouse has always had. + for i := 0; i < 2; i++ { + select { + case <-attempts: + case <-time.After(10 * time.Second): + t.Fatal("a node with nothing to rouse it stopped reconnecting") + } + } +} + +type saidSoFar struct { + mu sync.Mutex + said []string +} + +func (s *saidSoFar) say(line string) { + s.mu.Lock() + defer s.mu.Unlock() + s.said = append(s.said, line) +} + +func (s *saidSoFar) all() string { + s.mu.Lock() + defer s.mu.Unlock() + return strings.Join(s.said, "\n") +} diff --git a/internal/link/run.go b/internal/link/run.go index 091e2e2..dc62e5c 100644 --- a/internal/link/run.go +++ b/internal/link/run.go @@ -57,7 +57,39 @@ type Announce func(string) // hours. Retrying every second for hours is a node shouting into nothing; waiting a minute after // a broker blip is a node that is needlessly late. So it starts fast and slows down, and resets // once a connection has actually held. +// Roused is a channel that says the machine has reason to believe its link is stale — it woke +// from suspend, or its network changed. +// +// **The machine knows before any timeout does.** A suspended laptop's connection is dead the +// moment it wakes, and heartbeats find that out in twenty or thirty seconds; for that time the +// node believes it is in the mesh and is not, which is the one state this design says must never +// be indistinguishable from being connected. Nothing new listens on the node to arrange it — the +// signal a service manager already sends is enough (novox/hq ADR 0004). +// +// Nil is allowed and means nothing ever rouses it, which is every machine that does not suspend. +type Roused <-chan struct{} + func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeout time.Duration) error { + return HoldRoused(ctx, m, apply, say, timeout, nil) +} + +// HoldRoused is Hold, told when the machine has reason to think its link is stale. +func HoldRoused(ctx context.Context, m Membership, apply Applier, say Announce, + timeout time.Duration, roused Roused) error { + + return holdWith(ctx, func(ctx context.Context) error { + return Run(ctx, m, apply, say, timeout) + }, say, roused) +} + +// attempt is one try at holding the link open, returning when it ends for any reason. +// +// Named so the loop below can be driven without a broker. What the loop decides — when to wait, +// how long, what being roused does — is the part with the reasoning in it, and it was reachable +// only through a real connection before. +type attempt func(context.Context) error + +func holdWith(ctx context.Context, run attempt, say Announce, roused Roused) error { const ( first = 2 * time.Second most = 2 * time.Minute @@ -70,7 +102,31 @@ func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeou for { began := time.Now() - err := Run(ctx, m, apply, say, timeout) + + // The link runs under a context this loop can cancel, so being roused ends the current + // attempt rather than only shortening the wait after it. + // + // **That is the whole of it.** After a resume the socket looks perfectly healthy from + // inside this process — there is no error and no close, because nothing has tried to + // send anything. It is heartbeats that eventually discover it, twenty or thirty seconds + // later. A machine that knows it just woke does not have to wait to be told. + // + // A rouse that turns out to be spurious costs one reconnect, which is cheap and + // idempotent: the node redeclares its queue and anything unacknowledged is redelivered. + // The alternative costs half a minute of believing it is in a mesh it has left. + trying, done := context.WithCancel(ctx) + if roused != nil { + go func() { + select { + case <-trying.Done(): + case <-roused: + say("woken or moved — dropping the link and opening it again") + done() + } + }() + } + err := run(trying) + done() if ctx.Err() != nil { return nil } @@ -95,6 +151,14 @@ func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeou select { case <-ctx.Done(): return nil + case <-roused: + // And it does not serve out a wait computed for a broker that was restarting, either. + // + // The backoff is not *reset* by this. Being roused says the machine changed, not that + // whatever was refusing the connection has stopped — a laptop woken repeatedly on a + // network with no route would otherwise retry at full speed for as long as somebody + // keeps opening the lid. + say("woken or moved — trying again now") case <-time.After(wait): } if wait *= 2; wait > most { diff --git a/packaging/nox-mesh-host-network.sh b/packaging/nox-mesh-host-network.sh new file mode 100755 index 0000000..31eff2c --- /dev/null +++ b/packaging/nox-mesh-host-network.sh @@ -0,0 +1,22 @@ +#!/bin/sh +# Rouse the host when this machine's network changes. +# +# Installed as a NetworkManager dispatcher script (/etc/NetworkManager/dispatcher.d) and as a +# networkd-dispatcher one. Both hand the interface and the event as arguments; both are ignored +# beyond the event, because *which* interface changed does not matter — what matters is that a +# connection opened over the old route is now pointing at nothing, and that is true whichever +# interface it was. +# +# A machine that moves from wifi to ethernet holds a socket that looks perfectly healthy from +# inside the process: no error, no close, because nothing has tried to send anything. Heartbeats +# find it twenty or thirty seconds later. The machine knew immediately. +set -eu + +event="${2:-}" +case "$event" in + up|dhcp4-change|dhcp6-change|connectivity-change|routes-change) + # Only events that can change where packets go. `down` is deliberately not one: the link is + # already gone, reconnecting will fail, and the backoff exists for exactly that. + systemctl start --no-block nox-mesh-host-roused.service 2>/dev/null || true + ;; +esac diff --git a/packaging/nox-mesh-host-resume.service b/packaging/nox-mesh-host-resume.service new file mode 100644 index 0000000..bba7d2b --- /dev/null +++ b/packaging/nox-mesh-host-resume.service @@ -0,0 +1,14 @@ +# Rouse the host when this machine wakes. +# +# After sleep.target rather than before: the point is to act once the machine is back, and a +# signal sent on the way down would be read by a process that is about to be frozen with it. +[Unit] +Description=Rouse the Novox Mesh host after resume +After=suspend.target hibernate.target hybrid-sleep.target suspend-then-hibernate.target + +[Service] +Type=oneshot +ExecStart=/bin/sh -c 'systemctl start --no-block nox-mesh-host-roused.service' + +[Install] +WantedBy=suspend.target hibernate.target hybrid-sleep.target suspend-then-hibernate.target diff --git a/packaging/nox-mesh-host-roused.service b/packaging/nox-mesh-host-roused.service new file mode 100644 index 0000000..807cf77 --- /dev/null +++ b/packaging/nox-mesh-host-roused.service @@ -0,0 +1,17 @@ +# Tell the running host that this machine's link is probably stale. +# +# **A laptop knows it just woke; the link does not.** After a resume the socket looks perfectly +# healthy from inside the process — no error, no close, because nothing has tried to send +# anything. Heartbeats discover it twenty or thirty seconds later, and for that time the node +# believes it is in a mesh it has left. +# +# A signal rather than anything that listens: nothing may listen on a node (novox/hq ADR 0004), +# and a socket for this would be a control surface on every machine in exchange for saving twenty +# seconds. +[Unit] +Description=Tell the Novox Mesh host its link may be stale + +[Service] +Type=oneshot +# Nothing to do if the host is not running: this is a hint to a process, not a way to start one. +ExecStart=/bin/sh -c 'systemctl is-active --quiet nox-mesh-host.service && systemctl kill --signal=SIGHUP nox-mesh-host.service || true' diff --git a/packaging/roused_test.sh b/packaging/roused_test.sh new file mode 100755 index 0000000..00df906 --- /dev/null +++ b/packaging/roused_test.sh @@ -0,0 +1,42 @@ +#!/bin/sh +# The dispatcher script rouses on the events that change where packets go, and on nothing else. +# +# Checked by running it with a stub systemctl on PATH, because the thing worth testing is which +# events it acts on — a script that rouses on `down` would reconnect into a network that is gone, +# and one that rouses on nothing is the timeout it was written to avoid. +set -eu + +here=$(cd "$(dirname "$0")" && pwd) +work=$(mktemp -d) +trap 'rm -rf "$work"' EXIT + +mkdir -p "$work/bin" +cat > "$work/bin/systemctl" <<'STUB' +#!/bin/sh +echo "$@" >> "$ROUSED_LOG" +STUB +chmod +x "$work/bin/systemctl" +export PATH="$work/bin:$PATH" +export ROUSED_LOG="$work/rousings" +: > "$ROUSED_LOG" + +for event in up dhcp4-change connectivity-change routes-change; do + sh "$here/nox-mesh-host-network.sh" wlan0 "$event" +done +count=$(grep -c "nox-mesh-host-roused" "$ROUSED_LOG" || true) +if [ "$count" -ne 4 ]; then + echo "FAIL: four events that change where packets go roused $count time(s)" >&2 + exit 1 +fi + +: > "$ROUSED_LOG" +for event in down pre-up hostname ""; do + sh "$here/nox-mesh-host-network.sh" wlan0 "$event" +done +if [ -s "$ROUSED_LOG" ]; then + echo "FAIL: an event that does not change where packets go roused the host:" >&2 + cat "$ROUSED_LOG" >&2 + exit 1 +fi + +echo "the dispatcher rouses on route changes and nothing else"