Files
mesh-host/internal/link/run.go
T
jschoubben 4bff67ec69 A machine waiting to be enrolled is not a broken one
The launcher already ran `host run`, and `run` on a machine with no identity
exited with an error. So a freshly installed host, sitting exactly as intended
waiting for somebody to bring it a token, would have counted three failed
starts and rolled back its own installation.

It waits now, and says what it is waiting for. That is the *hosted* state from
the lifecycle: the host is running, it has no identity, and there is nobody to
link to. Every machine passes through it.

An identity that exists and cannot be read is still a fault rather than a wait.
Treating that as "not enrolled yet" would leave a node sitting quietly for ever
while the mesh believes it is a member.

Also: a node now says it is there once a minute. Nothing but its name, because
anything more would be a report, and reports are rare where this is constant --
reading one as the other would make a quiet node look like a stale one. Not
published mandatory, unlike a report: losing one is nothing, the next is a
minute away, and the mesh reads a gap rather than counting arrivals.

Verified in the lab: a node was stopped and the mesh said "out of touch 4m",
then it was started and the mesh said "here" again, without anything else being
touched.
2026-08-29 20:32:18 +02:00

273 lines
10 KiB
Go

package link
import (
"context"
"crypto/ed25519"
"encoding/json"
"errors"
"fmt"
"net/url"
"time"
amqp "github.com/rabbitmq/amqp091-go"
)
// ErrForged is what a node returns for a declaration whose signature is not the mesh's.
//
// Its own error, and it must never be confused with a malformed message. novox/hq ADR 0004
// requires a host to tell *this is not from the mesh I joined* apart from *this is malformed*:
// the first means somebody is trying, the second means something is broken.
var ErrForged = errors.New("this declaration was not signed by the mesh this node joined")
// AliveEvery is how often a node says it is there.
//
// Often enough that "no word for five minutes" means something, rarely enough that a hundred
// nodes are not a hundred messages a second. The mesh reads absence rather than presence, so what
// matters is the interval being known and steady.
const AliveEvery = 60 * time.Second
// Membership is what a node needs to reach its mesh again, held by the caller.
type Membership struct {
Node string
Broker string
Fingerprint string
Password string
Signer ed25519.PublicKey
}
// Applier is what the host does with a declaration that has been proved to come from the mesh.
//
// It receives the signature as well as the declaration, so the host can keep both: what it was
// told is kept signed and verified again when it is read back, which means the file on disk is
// trusted for the same reason the message was rather than for being local.
type Applier func(ctx context.Context, declaration, signature []byte) Report
// Announce is how the link says what is happening, so a node running unattended leaves an
// account of it. Nil is allowed and means say nothing.
type Announce func(string)
// Hold keeps this node in the mesh, reconnecting for as long as it is asked to.
//
// Disconnection is an ordinary situation and not a failure (novox/hq ADR 0004), so this does not
// give up. A laptop shut for a week comes back and reconnects; it does not come back needing
// somebody to start it again.
//
// The backoff exists because the two common reasons differ in how long they last: a broker
// restarting is back in seconds, and a machine that has moved to a network with no route may be
// hours. Retrying every second for hours is a node shouting into nothing; waiting a minute after
// a broker blip is a node that is needlessly late. So it starts fast and slows down, and resets
// once a connection has actually held.
func Hold(ctx context.Context, m Membership, apply Applier, say Announce, timeout time.Duration) error {
const (
first = 2 * time.Second
most = 2 * time.Minute
// A connection that lasted this long counts as having worked, so the next failure starts
// from the bottom again. Without it a node that reconnects and immediately drops climbs
// to the maximum and stays there, long after whatever caused it went away.
settled = 30 * time.Second
)
wait := first
for {
began := time.Now()
err := Run(ctx, m, apply, say, timeout)
if ctx.Err() != nil {
return nil
}
if time.Since(began) > settled {
wait = first
}
switch {
case errors.Is(err, ErrWrongCertificate):
// Said in full every time rather than folded into a retry count. This does not mean
// the network is down; it means what answered is not the mesh this node joined, and
// no amount of waiting fixes it. The node keeps running what it was last told, which
// is the right thing to do while somebody works out what happened.
say("the broker is not the one this node joined: " + err.Error())
say("this will not fix itself. This node keeps running what it was last told.")
case err != nil:
say(fmt.Sprintf("disconnected: %v — trying again in %s", err, wait))
default:
say(fmt.Sprintf("the link closed — trying again in %s", wait))
}
select {
case <-ctx.Done():
return nil
case <-time.After(wait):
}
if wait *= 2; wait > most {
wait = most
}
}
}
// Run holds the link open once, applying what arrives and reporting what happened.
//
// Outbound only, and nothing listens on this machine. Returns when the link ends, for any reason;
// Hold is what decides whether to open it again.
func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout time.Duration) error {
if say == nil {
say = func(string) {}
}
config, err := PinnedConfig(m.Fingerprint)
if err != nil {
return err
}
dsn := fmt.Sprintf("amqps://%s:%s@%s/",
url.QueryEscape(m.Node), url.QueryEscape(m.Password), m.Broker)
conn, err := amqp.DialConfig(dsn, amqp.Config{
TLSClientConfig: config,
Dial: amqp.DefaultDial(timeout),
// Kept short so a node that has silently lost its route notices, rather than holding a
// connection the broker forgot about and believing it is still in the mesh.
Heartbeat: 10 * time.Second,
})
if err != nil {
if errors.Is(err, ErrWrongCertificate) {
return err
}
return fmt.Errorf("cannot reach the broker at %s: %w", m.Broker, err)
}
defer conn.Close()
channel, err := conn.Channel()
if err != nil {
return err
}
defer channel.Close()
queue := QueueFor(m.Node)
if _, err := channel.QueueDeclare(queue, true, false, false, false, nil); err != nil {
return fmt.Errorf("cannot declare this node's queue %s: %w", queue, err)
}
// One at a time. A declaration is applied to a machine, and applying two at once would race
// on the same filesystem — so the broker holds the next one until this one is finished,
// where it survives a restart.
if err := channel.Qos(1, 0, false); err != nil {
return err
}
deliveries, err := channel.ConsumeWithContext(ctx, queue, "", false, false, false, false, nil)
if err != nil {
return err
}
// Said, because it is the event anybody watching actually wants. Without it a node logs
// every failure and nothing on success, so a log full of "trying again" and then silence
// reads as still broken when it means the opposite.
say("in the mesh, consuming " + queue)
// A word every so often, so the mesh can tell a node that is quiet from one that is gone.
// Cheap on purpose: it carries a name and nothing else, because anything more would be a
// report, and reports are rare where this is constant.
beat := time.NewTicker(AliveEvery)
defer beat.Stop()
publishAlive(ctx, channel, m, say, timeout)
closed := conn.NotifyClose(make(chan *amqp.Error, 1))
// Published mandatory, so the broker hands back anything it cannot route rather than
// dropping it. Without this a report goes to an exchange with no matching binding, the
// publisher is told nothing, and the mesh believes this node never answered while the node
// believes it did — which is what happened when `report` was left unbound on the other side.
returned := channel.NotifyReturn(make(chan amqp.Return, 4))
go func() {
for r := range returned {
say(fmt.Sprintf("the broker could not route this node's %s: %s (%d %s)",
r.RoutingKey, r.Exchange, r.ReplyCode, r.ReplyText))
}
}()
for {
select {
case <-ctx.Done():
return nil
case <-beat.C:
publishAlive(ctx, channel, m, say, timeout)
case reason := <-closed:
return fmt.Errorf("the link closed: %v", reason)
case delivery, ok := <-deliveries:
if !ok {
return errors.New("the broker stopped delivering")
}
report := handle(ctx, m, apply, delivery)
switch {
case report.Refused != "":
say("refused a declaration: " + report.Refused)
case len(report.Failed) > 0:
say(fmt.Sprintf("applied %d and failed: %v", len(report.Applied), report.Failed))
default:
say(fmt.Sprintf("applied %d resource(s)", len(report.Applied)))
}
publishReport(ctx, channel, m, report, say, timeout)
// Acknowledged after the report is published. A node that dies between applying and
// reporting leaves the declaration on the broker and applies it again on return,
// which is safe because applying is reconciliation — it converges rather than
// repeating.
_ = delivery.Ack(false)
}
}
}
func handle(ctx context.Context, m Membership, apply Applier, delivery amqp.Delivery) Report {
return handleBody(ctx, m, delivery.Body, apply)
}
// handleBody is the whole of deciding whether to trust a message, separated from the broker so it
// can be tested as the security check it is rather than as message plumbing.
func handleBody(ctx context.Context, m Membership, body []byte, apply Applier) Report {
var signed Signed
if err := json.Unmarshal(body, &signed); err != nil {
return Report{Node: m.Node, Refused: "this message is not a declaration: " + err.Error()}
}
// Before anything is read out of it, let alone applied. The host applies whatever the link
// delivers, so this check is the difference between the mesh changing this machine and
// anybody changing it.
if !ed25519.Verify(m.Signer, signed.Declaration, signed.Signature) {
return Report{Node: m.Node, Refused: ErrForged.Error()}
}
return apply(ctx, signed.Declaration, signed.Signature)
}
func publishReport(ctx context.Context, channel *amqp.Channel, m Membership, report Report,
say Announce, timeout time.Duration) {
report.Node = m.Node
body, err := json.Marshal(report)
if err != nil {
say("cannot encode this node's own report: " + err.Error())
return
}
publish, cancel := context.WithTimeout(ctx, timeout)
defer cancel()
// Said rather than swallowed. A report that fails to publish leaves the mesh believing this
// node never answered, while the node believes it did — and the two would go on disagreeing
// with nothing anywhere saying so. That shape of fault is the one this project keeps finding.
if err := channel.PublishWithContext(publish, Exchange, KeyReport, true, false,
amqp.Publishing{ContentType: "application/json", Body: body}); err != nil {
say(fmt.Sprintf("applied, and could not tell the mesh: %v", err))
}
}
// publishAlive says this node is here, and nothing else.
func publishAlive(ctx context.Context, channel *amqp.Channel, m Membership, say Announce,
timeout time.Duration) {
body, err := json.Marshal(Alive{Node: m.Node})
if err != nil {
return
}
publish, cancel := context.WithTimeout(ctx, timeout)
defer cancel()
// Not mandatory, unlike a report. Losing one is nothing: the next is a minute away, and the
// mesh is reading a gap rather than counting arrivals. Insisting on delivery would turn a
// harmless miss into a logged failure every minute.
if err := channel.PublishWithContext(publish, Exchange, KeyAlive, false, false,
amqp.Publishing{ContentType: "application/json", Body: body}); err != nil {
say("could not tell the mesh this node is here: " + err.Error())
}
}