A node that loses its mesh comes back on its own

Disconnection is an ordinary situation and not a failure, and until now the
host treated it as the end: the link dropped and the process returned. A laptop
shut for a week would have come back needing somebody to start it again.

Now it reconnects, with a backoff that starts at two seconds and slows to two
minutes. The two common reasons differ in how long they last -- a broker
restarting is back in seconds, a machine that has moved to a network with no
route may be hours -- so it starts fast and slows down, and resets once a
connection has actually held for thirty seconds. Without that reset, a node
that reconnects and immediately drops climbs to the maximum and stays there
long after the cause is gone.

A wrong certificate is said in full every time rather than folded into a retry
count. That does not mean the network is down; it means what answered is not
the mesh this node joined, and no waiting fixes it.

And it says when it gets back in. It logged every failure and nothing on
success, so a log full of "trying again" followed by silence read as still
broken when it meant the opposite.

The other half: a node now keeps what it was told, not only what it applied.
The record of what was applied holds an id, a type and a target -- what removal
needs, not what creation needs -- so it could not be re-applied. The
declaration is kept whole, signed, and verified again every time it is read
back, so the file on disk is trusted for the same reason the message was rather
than for being local. A tampered one is refused, and so is one signed by
another mesh.

With both, the host reconciles against what it was last told every five
minutes, connected or not. That is not polling for changes -- changes are
pushed -- it is the answer to a machine drifting: a file edited by hand, a
container somebody stopped, a service that died.

Verified in the lab. The broker was stopped: the node retried at 2s, 4s, 8s,
saying why each time, and kept its overlay up throughout. The broker came back
and the node rejoined without being touched. A declaration published while a
node was away was waiting on the broker and applied the moment it connected,
which is the buffer ADR 0006 describes doing its job.
This commit is contained in:
2026-08-29 20:17:49 +02:00
parent ba31eef80f
commit 980e12a850
5 changed files with 385 additions and 12 deletions
+65 -4
View File
@@ -546,16 +546,64 @@ func runLink(ctx context.Context, opts options) error {
fmt.Printf("node %s, linking to %s\n", mine.Node, mine.Membership.Broker)
apply := func(ctx context.Context, raw []byte) link.Report {
return applyDeclared(ctx, opts, raw)
apply := func(ctx context.Context, raw, signature []byte) link.Report {
return applyAndKeep(ctx, opts, raw, &store.Declared{Declaration: raw, Signature: signature})
}
return link.Run(ctx, link.Membership{
say := func(line string) { fmt.Println(line) }
// Two things at once, and the second is what makes disconnection ordinary. The link brings
// new declarations; this holds the machine in the last one whether the link is up or not. A
// laptop shut for a week comes back and reconciles — it does not come back and ask what it is
// (novox/hq ADR 0004).
go holdTheMachine(ctx, opts, mine, say)
return link.Hold(ctx, link.Membership{
Node: mine.Node,
Broker: mine.Membership.Broker,
Fingerprint: mine.Membership.Fingerprint,
Password: mine.Membership.Password,
Signer: mine.Membership.Signer,
}, apply, func(line string) { fmt.Println(line) }, opts.timeout)
}, apply, say, opts.timeout)
}
// ReconcileEvery is how often a node re-applies what it was last told.
//
// Not driven by the link. Changes are pushed, so this is not polling for them — it is the answer
// to a machine drifting: a file edited by hand, a container stopped by somebody, a service that
// died. A node that only acted when told would hold its state exactly until something else
// changed it, and then for ever.
const ReconcileEvery = 5 * time.Minute
func holdTheMachine(ctx context.Context, opts options, mine identity.Identity, say link.Announce) {
ticker := time.NewTicker(ReconcileEvery)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
return
case <-ticker.C:
}
declared, err := store.LoadDeclared(store.DeclaredPath(opts.state), mine.Membership.Signer)
if errors.Is(err, store.ErrNothingDeclared) {
// Nothing to hold this machine to yet. Ordinary on a node that has enrolled and not
// been assigned anything.
continue
}
if err != nil {
say("cannot re-apply what this node was told: " + err.Error())
continue
}
report := applyDeclared(ctx, opts, declared)
switch {
case report.Refused != "":
say("what this node was last told no longer applies: " + report.Refused)
case len(report.Failed) > 0:
say(fmt.Sprintf("holding this machine: %d applied, and %v", len(report.Applied), report.Failed))
}
}
}
// applyDeclared applies a declaration that has already been proved to come from the mesh.
@@ -564,6 +612,12 @@ func runLink(ctx context.Context, opts options) error {
// the question "is this from the mesh I joined" is settled — which is why this can treat the
// bytes as instructions.
func applyDeclared(ctx context.Context, opts options, raw []byte) link.Report {
return applyAndKeep(ctx, opts, raw, nil)
}
// applyAndKeep applies a declaration and, when it came from the mesh, keeps it so this node can
// go on obeying it while disconnected.
func applyAndKeep(ctx context.Context, opts options, raw []byte, signed *store.Declared) link.Report {
declared, err := declaration.Parse(raw)
if err != nil {
return link.Report{Refused: err.Error()}
@@ -602,6 +656,13 @@ func applyDeclared(ctx context.Context, opts options, raw []byte) link.Report {
for _, change := range outcome.Outcomes {
report.Applied = append(report.Applied, change.ID)
}
// Kept whichever way it went, so a node that is disconnected next minute still knows what it
// was told. Saved after applying rather than before: what is kept is what this node acted on.
if signed != nil {
if err := store.SaveDeclared(store.DeclaredPath(opts.state), *signed); err != nil {
report.Failed = map[string]string{"keeping the declaration": err.Error()}
}
}
if applyErr != nil {
report.Failed = map[string]string{"apply": applyErr.Error()}
}