diff --git a/Makefile b/Makefile index e865709..ebe96e7 100644 --- a/Makefile +++ b/Makefile @@ -23,6 +23,7 @@ hosts: packaging-test: @./packaging/rollback_test.sh @./packaging/launch_test.sh + @./packaging/roused_test.sh fmt: @test -z "$$(gofmt -l . )" || { echo "unformatted:"; gofmt -l . ; exit 1; } diff --git a/cmd/mesh-host/main.go b/cmd/mesh-host/main.go index 2eb9eac..6fc4d63 100644 --- a/cmd/mesh-host/main.go +++ b/cmd/mesh-host/main.go @@ -602,13 +602,49 @@ func runLink(ctx context.Context, opts options) error { // (novox/hq ADR 0004). go holdTheMachine(ctx, opts, mine, say) - return link.Hold(ctx, link.Membership{ + return link.HoldRoused(ctx, link.Membership{ Node: mine.Node, Broker: mine.Membership.Broker, Fingerprint: mine.Membership.Fingerprint, Password: mine.Membership.Password, Signer: mine.Membership.Signer, - }, apply, say, opts.timeout) + }, apply, say, opts.timeout, rousedBySignal(ctx)) +} + +// rousedBySignal is the machine telling this process that its link is probably stale. +// +// **A signal, because nothing may listen on a node** (novox/hq ADR 0004). A socket for this would +// be a control surface on every machine, reachable by anything that can reach the machine, in +// exchange for saving twenty seconds — and the whole security argument rests on there not being +// one. A signal is delivered by the service manager to a process it already supervises. +// +// SIGHUP, because that is the signal a long-running program conventionally reads as *look again*, +// and nothing here is being reloaded from a file that a different signal would suit better. +// +// Dropped rather than queued when one arrives while another is unread: two wakes in the same +// instant are one wake, and a machine that suspends and resumes repeatedly must not build a +// backlog of reconnections to work through. +func rousedBySignal(ctx context.Context) link.Roused { + woken := make(chan os.Signal, 1) + signal.Notify(woken, syscall.SIGHUP) + + out := make(chan struct{}, 1) + go func() { + defer signal.Stop(woken) + for { + select { + case <-ctx.Done(): + return + case <-woken: + select { + case out <- struct{}{}: + default: + // One is already waiting to be read. Two wakes in the same instant are one. + } + } + } + }() + return out } // ReconcileEvery is how often a node re-applies what it was last told. diff --git a/internal/link/roused_test.go b/internal/link/roused_test.go new file mode 100644 index 0000000..1071566 --- /dev/null +++ b/internal/link/roused_test.go @@ -0,0 +1,103 @@ +package link + +import ( + "context" + "errors" + "strings" + "sync" + "testing" + "time" +) + +// A machine that just woke does not wait to be told its link is dead. +// +// After a resume the socket looks perfectly healthy from inside the process — no error, no close, +// because nothing has tried to send anything. Heartbeats discover it twenty or thirty seconds +// later, and for that time the node believes it is in a mesh it has left, which is the one state +// this design says must never be indistinguishable from being connected. +func TestBeingRousedEndsTheCurrentAttemptRatherThanWaitingForATimeout(t *testing.T) { + // A link that never returns on its own, which is exactly what a suspended connection is. + held := make(chan context.Context, 4) + running := func(ctx context.Context) error { + held <- ctx + <-ctx.Done() + return errors.New("the link ended") + } + + rouse := make(chan struct{}, 1) + said := &saidSoFar{} + ctx, stop := context.WithCancel(context.Background()) + defer stop() + + finished := make(chan error, 1) + go func() { finished <- holdWith(ctx, running, said.say, rouse) }() + + first := <-held + select { + case <-first.Done(): + t.Fatal("the link ended before anything roused it") + case <-time.After(50 * time.Millisecond): + } + + rouse <- struct{}{} + select { + case <-first.Done(): + case <-time.After(2 * time.Second): + t.Fatal("the machine woke and the link was left running against a socket that is gone") + } + + // And it opens another one rather than stopping. + select { + case <-held: + case <-time.After(5 * time.Second): + t.Fatal("the link was dropped and never opened again") + } + if !strings.Contains(said.all(), "woken or moved") { + t.Fatalf("nothing was said about why the link was dropped:\n%s", said.all()) + } + + stop() + select { + case <-finished: + case <-time.After(2 * time.Second): + t.Fatal("it did not stop when asked") + } +} + +// A machine nothing ever rouses is every machine that does not suspend, and must behave as before. +func TestAMachineNothingRousesIsUnaffected(t *testing.T) { + attempts := make(chan struct{}, 4) + running := func(context.Context) error { + attempts <- struct{}{} + return errors.New("the link ended") + } + ctx, stop := context.WithCancel(context.Background()) + defer stop() + go func() { _ = holdWith(ctx, running, func(string) {}, nil) }() + + // It keeps trying, which is the behaviour a node with no rouse has always had. + for i := 0; i < 2; i++ { + select { + case <-attempts: + case <-time.After(10 * time.Second): + t.Fatal("a node with nothing to rouse it stopped reconnecting") + } + } +} + +type saidSoFar struct { + mu sync.Mutex + said []string +} + +func (s *saidSoFar) say(line string) { + s.mu.Lock() + defer s.mu.Unlock() + s.said = append(s.said, line) +} + +func (s *saidSoFar) all() string { + s.mu.Lock() + defer s.mu.Unlock() + return strings.Join(s.said, "\n") +} diff --git a/internal/link/run.go b/internal/link/run.go index 091e2e2..dc62e5c 100644 --- a/internal/link/run.go +++ b/internal/link/run.go @@ -57,7 +57,39 @@ type Announce func(string) // hours. Retrying every second for hours is a node shouting into nothing; waiting a minute after // a broker blip is a node that is needlessly late. So it starts fast and slows down, and resets // once a connection has actually held. +// Roused is a channel that says the machine has reason to believe its link is stale — it woke +// from suspend, or its network changed. +// +// **The machine knows before any timeout does.** A suspended laptop's connection is dead the +// moment it wakes, and heartbeats find that out in twenty or thirty seconds; for that time the +// node believes it is in the mesh and is not, which is the one state this design says must never +// be indistinguishable from being connected. Nothing new listens on the node to arrange it — the +// signal a service manager already sends is enough (novox/hq ADR 0004). +// +// Nil is allowed and means nothing ever rouses it, which is every machine that does not suspend. +type Roused <-chan struct{} + func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeout time.Duration) error { + return HoldRoused(ctx, m, apply, say, timeout, nil) +} + +// HoldRoused is Hold, told when the machine has reason to think its link is stale. +func HoldRoused(ctx context.Context, m Membership, apply Applier, say Announce, + timeout time.Duration, roused Roused) error { + + return holdWith(ctx, func(ctx context.Context) error { + return Run(ctx, m, apply, say, timeout) + }, say, roused) +} + +// attempt is one try at holding the link open, returning when it ends for any reason. +// +// Named so the loop below can be driven without a broker. What the loop decides — when to wait, +// how long, what being roused does — is the part with the reasoning in it, and it was reachable +// only through a real connection before. +type attempt func(context.Context) error + +func holdWith(ctx context.Context, run attempt, say Announce, roused Roused) error { const ( first = 2 * time.Second most = 2 * time.Minute @@ -70,7 +102,31 @@ func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeou for { began := time.Now() - err := Run(ctx, m, apply, say, timeout) + + // The link runs under a context this loop can cancel, so being roused ends the current + // attempt rather than only shortening the wait after it. + // + // **That is the whole of it.** After a resume the socket looks perfectly healthy from + // inside this process — there is no error and no close, because nothing has tried to + // send anything. It is heartbeats that eventually discover it, twenty or thirty seconds + // later. A machine that knows it just woke does not have to wait to be told. + // + // A rouse that turns out to be spurious costs one reconnect, which is cheap and + // idempotent: the node redeclares its queue and anything unacknowledged is redelivered. + // The alternative costs half a minute of believing it is in a mesh it has left. + trying, done := context.WithCancel(ctx) + if roused != nil { + go func() { + select { + case <-trying.Done(): + case <-roused: + say("woken or moved — dropping the link and opening it again") + done() + } + }() + } + err := run(trying) + done() if ctx.Err() != nil { return nil } @@ -95,6 +151,14 @@ func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeou select { case <-ctx.Done(): return nil + case <-roused: + // And it does not serve out a wait computed for a broker that was restarting, either. + // + // The backoff is not *reset* by this. Being roused says the machine changed, not that + // whatever was refusing the connection has stopped — a laptop woken repeatedly on a + // network with no route would otherwise retry at full speed for as long as somebody + // keeps opening the lid. + say("woken or moved — trying again now") case <-time.After(wait): } if wait *= 2; wait > most { diff --git a/packaging/nox-mesh-host-network.sh b/packaging/nox-mesh-host-network.sh new file mode 100755 index 0000000..31eff2c --- /dev/null +++ b/packaging/nox-mesh-host-network.sh @@ -0,0 +1,22 @@ +#!/bin/sh +# Rouse the host when this machine's network changes. +# +# Installed as a NetworkManager dispatcher script (/etc/NetworkManager/dispatcher.d) and as a +# networkd-dispatcher one. Both hand the interface and the event as arguments; both are ignored +# beyond the event, because *which* interface changed does not matter — what matters is that a +# connection opened over the old route is now pointing at nothing, and that is true whichever +# interface it was. +# +# A machine that moves from wifi to ethernet holds a socket that looks perfectly healthy from +# inside the process: no error, no close, because nothing has tried to send anything. Heartbeats +# find it twenty or thirty seconds later. The machine knew immediately. +set -eu + +event="${2:-}" +case "$event" in + up|dhcp4-change|dhcp6-change|connectivity-change|routes-change) + # Only events that can change where packets go. `down` is deliberately not one: the link is + # already gone, reconnecting will fail, and the backoff exists for exactly that. + systemctl start --no-block nox-mesh-host-roused.service 2>/dev/null || true + ;; +esac diff --git a/packaging/nox-mesh-host-resume.service b/packaging/nox-mesh-host-resume.service new file mode 100644 index 0000000..bba7d2b --- /dev/null +++ b/packaging/nox-mesh-host-resume.service @@ -0,0 +1,14 @@ +# Rouse the host when this machine wakes. +# +# After sleep.target rather than before: the point is to act once the machine is back, and a +# signal sent on the way down would be read by a process that is about to be frozen with it. +[Unit] +Description=Rouse the Novox Mesh host after resume +After=suspend.target hibernate.target hybrid-sleep.target suspend-then-hibernate.target + +[Service] +Type=oneshot +ExecStart=/bin/sh -c 'systemctl start --no-block nox-mesh-host-roused.service' + +[Install] +WantedBy=suspend.target hibernate.target hybrid-sleep.target suspend-then-hibernate.target diff --git a/packaging/nox-mesh-host-roused.service b/packaging/nox-mesh-host-roused.service new file mode 100644 index 0000000..807cf77 --- /dev/null +++ b/packaging/nox-mesh-host-roused.service @@ -0,0 +1,17 @@ +# Tell the running host that this machine's link is probably stale. +# +# **A laptop knows it just woke; the link does not.** After a resume the socket looks perfectly +# healthy from inside the process — no error, no close, because nothing has tried to send +# anything. Heartbeats discover it twenty or thirty seconds later, and for that time the node +# believes it is in a mesh it has left. +# +# A signal rather than anything that listens: nothing may listen on a node (novox/hq ADR 0004), +# and a socket for this would be a control surface on every machine in exchange for saving twenty +# seconds. +[Unit] +Description=Tell the Novox Mesh host its link may be stale + +[Service] +Type=oneshot +# Nothing to do if the host is not running: this is a hint to a process, not a way to start one. +ExecStart=/bin/sh -c 'systemctl is-active --quiet nox-mesh-host.service && systemctl kill --signal=SIGHUP nox-mesh-host.service || true' diff --git a/packaging/roused_test.sh b/packaging/roused_test.sh new file mode 100755 index 0000000..00df906 --- /dev/null +++ b/packaging/roused_test.sh @@ -0,0 +1,42 @@ +#!/bin/sh +# The dispatcher script rouses on the events that change where packets go, and on nothing else. +# +# Checked by running it with a stub systemctl on PATH, because the thing worth testing is which +# events it acts on — a script that rouses on `down` would reconnect into a network that is gone, +# and one that rouses on nothing is the timeout it was written to avoid. +set -eu + +here=$(cd "$(dirname "$0")" && pwd) +work=$(mktemp -d) +trap 'rm -rf "$work"' EXIT + +mkdir -p "$work/bin" +cat > "$work/bin/systemctl" <<'STUB' +#!/bin/sh +echo "$@" >> "$ROUSED_LOG" +STUB +chmod +x "$work/bin/systemctl" +export PATH="$work/bin:$PATH" +export ROUSED_LOG="$work/rousings" +: > "$ROUSED_LOG" + +for event in up dhcp4-change connectivity-change routes-change; do + sh "$here/nox-mesh-host-network.sh" wlan0 "$event" +done +count=$(grep -c "nox-mesh-host-roused" "$ROUSED_LOG" || true) +if [ "$count" -ne 4 ]; then + echo "FAIL: four events that change where packets go roused $count time(s)" >&2 + exit 1 +fi + +: > "$ROUSED_LOG" +for event in down pre-up hostname ""; do + sh "$here/nox-mesh-host-network.sh" wlan0 "$event" +done +if [ -s "$ROUSED_LOG" ]; then + echo "FAIL: an event that does not change where packets go roused the host:" >&2 + cat "$ROUSED_LOG" >&2 + exit 1 +fi + +echo "the dispatcher rouses on route changes and nothing else"