A machine that wakes or moves says so, instead of waiting to be told

A suspended laptop's connection is dead the moment it wakes, and the socket
looks perfectly healthy from inside the process — no error, no close, because
nothing has tried to send anything. Heartbeats find out twenty or thirty
seconds later. For that time the node believes it is in a mesh it has left,
which is the one state this design says must never be indistinguishable from
being connected. The machine knew immediately.

So being roused ends the current attempt rather than only shortening the wait
after it: shortening the wait would do nothing at all, because the process is
not waiting — it is sitting inside a connection that will not return.

A signal, because nothing may listen on a node (novox/hq ADR 0004). A socket
for this would be a control surface on every machine, reachable by anything
that can reach the machine, in exchange for saving twenty seconds — and the
whole security argument rests on there not being one.

Two rouses in the same instant are one: a machine suspending and resuming
repeatedly must not build a backlog of reconnections to work through. And the
backoff is not reset by being roused — that says the machine changed, not that
whatever was refusing the connection has stopped, and a laptop woken on a
network with no route would otherwise retry at full speed for as long as
somebody keeps opening the lid.

The dispatcher acts on the events that change where packets go and not on
`down`: the link is already gone there, reconnecting will fail, and the backoff
exists for exactly that.
This commit is contained in:
2026-08-31 10:21:26 +02:00
parent 5bc0006e83
commit 8fcfa88fe0
8 changed files with 302 additions and 3 deletions
+1
View File
@@ -23,6 +23,7 @@ hosts:
packaging-test: packaging-test:
@./packaging/rollback_test.sh @./packaging/rollback_test.sh
@./packaging/launch_test.sh @./packaging/launch_test.sh
@./packaging/roused_test.sh
fmt: fmt:
@test -z "$$(gofmt -l . )" || { echo "unformatted:"; gofmt -l . ; exit 1; } @test -z "$$(gofmt -l . )" || { echo "unformatted:"; gofmt -l . ; exit 1; }
+38 -2
View File
@@ -602,13 +602,49 @@ func runLink(ctx context.Context, opts options) error {
// (novox/hq ADR 0004). // (novox/hq ADR 0004).
go holdTheMachine(ctx, opts, mine, say) go holdTheMachine(ctx, opts, mine, say)
return link.Hold(ctx, link.Membership{ return link.HoldRoused(ctx, link.Membership{
Node: mine.Node, Node: mine.Node,
Broker: mine.Membership.Broker, Broker: mine.Membership.Broker,
Fingerprint: mine.Membership.Fingerprint, Fingerprint: mine.Membership.Fingerprint,
Password: mine.Membership.Password, Password: mine.Membership.Password,
Signer: mine.Membership.Signer, Signer: mine.Membership.Signer,
}, apply, say, opts.timeout) }, apply, say, opts.timeout, rousedBySignal(ctx))
}
// rousedBySignal is the machine telling this process that its link is probably stale.
//
// **A signal, because nothing may listen on a node** (novox/hq ADR 0004). A socket for this would
// be a control surface on every machine, reachable by anything that can reach the machine, in
// exchange for saving twenty seconds — and the whole security argument rests on there not being
// one. A signal is delivered by the service manager to a process it already supervises.
//
// SIGHUP, because that is the signal a long-running program conventionally reads as *look again*,
// and nothing here is being reloaded from a file that a different signal would suit better.
//
// Dropped rather than queued when one arrives while another is unread: two wakes in the same
// instant are one wake, and a machine that suspends and resumes repeatedly must not build a
// backlog of reconnections to work through.
func rousedBySignal(ctx context.Context) link.Roused {
woken := make(chan os.Signal, 1)
signal.Notify(woken, syscall.SIGHUP)
out := make(chan struct{}, 1)
go func() {
defer signal.Stop(woken)
for {
select {
case <-ctx.Done():
return
case <-woken:
select {
case out <- struct{}{}:
default:
// One is already waiting to be read. Two wakes in the same instant are one.
}
}
}
}()
return out
} }
// ReconcileEvery is how often a node re-applies what it was last told. // ReconcileEvery is how often a node re-applies what it was last told.
+103
View File
@@ -0,0 +1,103 @@
package link
import (
"context"
"errors"
"strings"
"sync"
"testing"
"time"
)
// A machine that just woke does not wait to be told its link is dead.
//
// After a resume the socket looks perfectly healthy from inside the process — no error, no close,
// because nothing has tried to send anything. Heartbeats discover it twenty or thirty seconds
// later, and for that time the node believes it is in a mesh it has left, which is the one state
// this design says must never be indistinguishable from being connected.
func TestBeingRousedEndsTheCurrentAttemptRatherThanWaitingForATimeout(t *testing.T) {
// A link that never returns on its own, which is exactly what a suspended connection is.
held := make(chan context.Context, 4)
running := func(ctx context.Context) error {
held <- ctx
<-ctx.Done()
return errors.New("the link ended")
}
rouse := make(chan struct{}, 1)
said := &saidSoFar{}
ctx, stop := context.WithCancel(context.Background())
defer stop()
finished := make(chan error, 1)
go func() { finished <- holdWith(ctx, running, said.say, rouse) }()
first := <-held
select {
case <-first.Done():
t.Fatal("the link ended before anything roused it")
case <-time.After(50 * time.Millisecond):
}
rouse <- struct{}{}
select {
case <-first.Done():
case <-time.After(2 * time.Second):
t.Fatal("the machine woke and the link was left running against a socket that is gone")
}
// And it opens another one rather than stopping.
select {
case <-held:
case <-time.After(5 * time.Second):
t.Fatal("the link was dropped and never opened again")
}
if !strings.Contains(said.all(), "woken or moved") {
t.Fatalf("nothing was said about why the link was dropped:\n%s", said.all())
}
stop()
select {
case <-finished:
case <-time.After(2 * time.Second):
t.Fatal("it did not stop when asked")
}
}
// A machine nothing ever rouses is every machine that does not suspend, and must behave as before.
func TestAMachineNothingRousesIsUnaffected(t *testing.T) {
attempts := make(chan struct{}, 4)
running := func(context.Context) error {
attempts <- struct{}{}
return errors.New("the link ended")
}
ctx, stop := context.WithCancel(context.Background())
defer stop()
go func() { _ = holdWith(ctx, running, func(string) {}, nil) }()
// It keeps trying, which is the behaviour a node with no rouse has always had.
for i := 0; i < 2; i++ {
select {
case <-attempts:
case <-time.After(10 * time.Second):
t.Fatal("a node with nothing to rouse it stopped reconnecting")
}
}
}
type saidSoFar struct {
mu sync.Mutex
said []string
}
func (s *saidSoFar) say(line string) {
s.mu.Lock()
defer s.mu.Unlock()
s.said = append(s.said, line)
}
func (s *saidSoFar) all() string {
s.mu.Lock()
defer s.mu.Unlock()
return strings.Join(s.said, "\n")
}
+65 -1
View File
@@ -57,7 +57,39 @@ type Announce func(string)
// hours. Retrying every second for hours is a node shouting into nothing; waiting a minute after // hours. Retrying every second for hours is a node shouting into nothing; waiting a minute after
// a broker blip is a node that is needlessly late. So it starts fast and slows down, and resets // a broker blip is a node that is needlessly late. So it starts fast and slows down, and resets
// once a connection has actually held. // once a connection has actually held.
// Roused is a channel that says the machine has reason to believe its link is stale — it woke
// from suspend, or its network changed.
//
// **The machine knows before any timeout does.** A suspended laptop's connection is dead the
// moment it wakes, and heartbeats find that out in twenty or thirty seconds; for that time the
// node believes it is in the mesh and is not, which is the one state this design says must never
// be indistinguishable from being connected. Nothing new listens on the node to arrange it — the
// signal a service manager already sends is enough (novox/hq ADR 0004).
//
// Nil is allowed and means nothing ever rouses it, which is every machine that does not suspend.
type Roused <-chan struct{}
func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeout time.Duration) error { func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeout time.Duration) error {
return HoldRoused(ctx, m, apply, say, timeout, nil)
}
// HoldRoused is Hold, told when the machine has reason to think its link is stale.
func HoldRoused(ctx context.Context, m Membership, apply Applier, say Announce,
timeout time.Duration, roused Roused) error {
return holdWith(ctx, func(ctx context.Context) error {
return Run(ctx, m, apply, say, timeout)
}, say, roused)
}
// attempt is one try at holding the link open, returning when it ends for any reason.
//
// Named so the loop below can be driven without a broker. What the loop decides — when to wait,
// how long, what being roused does — is the part with the reasoning in it, and it was reachable
// only through a real connection before.
type attempt func(context.Context) error
func holdWith(ctx context.Context, run attempt, say Announce, roused Roused) error {
const ( const (
first = 2 * time.Second first = 2 * time.Second
most = 2 * time.Minute most = 2 * time.Minute
@@ -70,7 +102,31 @@ func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeou
for { for {
began := time.Now() began := time.Now()
err := Run(ctx, m, apply, say, timeout)
// The link runs under a context this loop can cancel, so being roused ends the current
// attempt rather than only shortening the wait after it.
//
// **That is the whole of it.** After a resume the socket looks perfectly healthy from
// inside this process — there is no error and no close, because nothing has tried to
// send anything. It is heartbeats that eventually discover it, twenty or thirty seconds
// later. A machine that knows it just woke does not have to wait to be told.
//
// A rouse that turns out to be spurious costs one reconnect, which is cheap and
// idempotent: the node redeclares its queue and anything unacknowledged is redelivered.
// The alternative costs half a minute of believing it is in a mesh it has left.
trying, done := context.WithCancel(ctx)
if roused != nil {
go func() {
select {
case <-trying.Done():
case <-roused:
say("woken or moved — dropping the link and opening it again")
done()
}
}()
}
err := run(trying)
done()
if ctx.Err() != nil { if ctx.Err() != nil {
return nil return nil
} }
@@ -95,6 +151,14 @@ func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeou
select { select {
case <-ctx.Done(): case <-ctx.Done():
return nil return nil
case <-roused:
// And it does not serve out a wait computed for a broker that was restarting, either.
//
// The backoff is not *reset* by this. Being roused says the machine changed, not that
// whatever was refusing the connection has stopped — a laptop woken repeatedly on a
// network with no route would otherwise retry at full speed for as long as somebody
// keeps opening the lid.
say("woken or moved — trying again now")
case <-time.After(wait): case <-time.After(wait):
} }
if wait *= 2; wait > most { if wait *= 2; wait > most {
+22
View File
@@ -0,0 +1,22 @@
#!/bin/sh
# Rouse the host when this machine's network changes.
#
# Installed as a NetworkManager dispatcher script (/etc/NetworkManager/dispatcher.d) and as a
# networkd-dispatcher one. Both hand the interface and the event as arguments; both are ignored
# beyond the event, because *which* interface changed does not matter — what matters is that a
# connection opened over the old route is now pointing at nothing, and that is true whichever
# interface it was.
#
# A machine that moves from wifi to ethernet holds a socket that looks perfectly healthy from
# inside the process: no error, no close, because nothing has tried to send anything. Heartbeats
# find it twenty or thirty seconds later. The machine knew immediately.
set -eu
event="${2:-}"
case "$event" in
up|dhcp4-change|dhcp6-change|connectivity-change|routes-change)
# Only events that can change where packets go. `down` is deliberately not one: the link is
# already gone, reconnecting will fail, and the backoff exists for exactly that.
systemctl start --no-block nox-mesh-host-roused.service 2>/dev/null || true
;;
esac
+14
View File
@@ -0,0 +1,14 @@
# Rouse the host when this machine wakes.
#
# After sleep.target rather than before: the point is to act once the machine is back, and a
# signal sent on the way down would be read by a process that is about to be frozen with it.
[Unit]
Description=Rouse the Novox Mesh host after resume
After=suspend.target hibernate.target hybrid-sleep.target suspend-then-hibernate.target
[Service]
Type=oneshot
ExecStart=/bin/sh -c 'systemctl start --no-block nox-mesh-host-roused.service'
[Install]
WantedBy=suspend.target hibernate.target hybrid-sleep.target suspend-then-hibernate.target
+17
View File
@@ -0,0 +1,17 @@
# Tell the running host that this machine's link is probably stale.
#
# **A laptop knows it just woke; the link does not.** After a resume the socket looks perfectly
# healthy from inside the process — no error, no close, because nothing has tried to send
# anything. Heartbeats discover it twenty or thirty seconds later, and for that time the node
# believes it is in a mesh it has left.
#
# A signal rather than anything that listens: nothing may listen on a node (novox/hq ADR 0004),
# and a socket for this would be a control surface on every machine in exchange for saving twenty
# seconds.
[Unit]
Description=Tell the Novox Mesh host its link may be stale
[Service]
Type=oneshot
# Nothing to do if the host is not running: this is a hint to a process, not a way to start one.
ExecStart=/bin/sh -c 'systemctl is-active --quiet nox-mesh-host.service && systemctl kill --signal=SIGHUP nox-mesh-host.service || true'
+42
View File
@@ -0,0 +1,42 @@
#!/bin/sh
# The dispatcher script rouses on the events that change where packets go, and on nothing else.
#
# Checked by running it with a stub systemctl on PATH, because the thing worth testing is which
# events it acts on — a script that rouses on `down` would reconnect into a network that is gone,
# and one that rouses on nothing is the timeout it was written to avoid.
set -eu
here=$(cd "$(dirname "$0")" && pwd)
work=$(mktemp -d)
trap 'rm -rf "$work"' EXIT
mkdir -p "$work/bin"
cat > "$work/bin/systemctl" <<'STUB'
#!/bin/sh
echo "$@" >> "$ROUSED_LOG"
STUB
chmod +x "$work/bin/systemctl"
export PATH="$work/bin:$PATH"
export ROUSED_LOG="$work/rousings"
: > "$ROUSED_LOG"
for event in up dhcp4-change connectivity-change routes-change; do
sh "$here/nox-mesh-host-network.sh" wlan0 "$event"
done
count=$(grep -c "nox-mesh-host-roused" "$ROUSED_LOG" || true)
if [ "$count" -ne 4 ]; then
echo "FAIL: four events that change where packets go roused $count time(s)" >&2
exit 1
fi
: > "$ROUSED_LOG"
for event in down pre-up hostname ""; do
sh "$here/nox-mesh-host-network.sh" wlan0 "$event"
done
if [ -s "$ROUSED_LOG" ]; then
echo "FAIL: an event that does not change where packets go roused the host:" >&2
cat "$ROUSED_LOG" >&2
exit 1
fi
echo "the dispatcher rouses on route changes and nothing else"