Files
mesh-controller/internal/link/enrolment.go
T
jschoubben 9681b288aa Keep what each machine did, so status can say what is wrong
A node reports back after applying a declaration: it worked, some of it
failed, or the whole thing was refused. A refusal or a failure moved
last_seen and the reason went to a log line — so "which machine is not
doing what it was told" had no answer the next morning, which is the
question a mesh exists to answer.

Refused and failed are kept as different things, because they are
different situations with different remedies: refused means the machine
is exactly as it was and what is wrong is in what was sent; failed means
it is in a state nobody declared and what is wrong is on the machine. One
word for both would make the record say less than the node did.

One row per node, replaced. The question is the machine's current state —
"this failed an hour ago and then succeeded" is not a machine anybody
needs to look at, and a table of every report would bury the ones that
matter under the ones that do not.

`status` now answers three questions in the order somebody asks them: is
anything broken, is anything not answering, is anything out of date. The
first has consequences now, the third is a plan for later, and a status
leading with the third would bury the first. A machine that has never
spoken is reported as quiet rather than as broken — new, switched off and
unreachable are not the same as tried and could not.

The mapping from a report to an outcome had no test at all, which the
injection caught: it is the code deciding which of those situations a
machine is in. It has four now, including that a partial report never
becomes the account of what the machine holds — the fault that destroyed
a substrate once.
2026-08-30 18:08:59 +02:00

176 lines
6.8 KiB
Go

package link
import (
"context"
"crypto/ed25519"
"crypto/rand"
"encoding/base64"
"errors"
"fmt"
"sort"
"github.com/novox/mesh-control/internal/broker"
"github.com/novox/mesh-control/internal/identity"
"github.com/novox/mesh-control/internal/inventory"
)
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
// recorded, and the node gets its own queue.
//
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
// — it holds both grants and asks each for its part, which is what the process running them is
// for.
type Enrolment struct {
Inventory *inventory.Inventory
Identity *identity.Identity
Management *broker.Management
Broker broker.Broker
}
// Enrol spends the token and records what the node presented.
//
// Order matters and it is the order things become irreversible. The token is spent first, in a
// single statement that both finds and marks it, so two machines racing on one secret produce one
// winner. Only then is a key recorded — because recording a key for a node whose token turned out
// to be spent would leave the mesh believing a machine that never had the right to join.
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (EnrolReply, error) {
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
if len(public) != ed25519.PublicKeySize {
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
len(public), ed25519.PublicKeySize)
}
node, err := e.Inventory.Redeem(ctx, secret)
if err != nil {
return EnrolReply{}, err
}
// From here the token is gone whatever happens next, so anything that fails leaves a node
// record with no live key — which is visible and fixable with a new token, where a spent
// token believed to be unspent is neither.
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and the key could not be recorded, so %s has no identity and "+
"needs a new token: %w", node.Name, err)
}
key, err := e.Identity.Active(ctx)
if err != nil {
return EnrolReply{}, err
}
reply := EnrolReply{
Accepted: true,
Node: node.Name,
Queue: QueueFor(node.Name),
Broker: e.Broker.Address,
Fingerprint: e.Broker.Fingerprint,
Signer: key.Public,
}
// The token's secret was the broker password up to this moment, which is what let this
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
// credential the node keeps for years is not the one that was pasted into a terminal.
if e.Management != nil {
password, err := freshPassword()
if err != nil {
return EnrolReply{}, err
}
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's broker password could not be replaced: %w",
node.Name, err)
}
reply.Password = password
}
// Recorded before the profile because the overlay is the first declaration this node will
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
// briefly as possible.
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
// stronger: without it the mesh cannot send this node a credential at all, and a node that
// enrolled without one will be refused a sealed file rather than quietly given none.
if request.SealingKey != "" {
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's sealing key could not be recorded: %w",
node.Name, err)
}
}
if request.OverlayKey != "" {
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's overlay key could not be recorded: %w", node.Name, err)
}
}
if profile != nil {
// Not fatal if it fails. The profile is what the control plane needs in order to decide
// what this machine should run, and it is reported again on every connection — so losing
// it here costs a decision that can be made later, not the enrolment.
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
}
return reply, nil
}
// freshPassword is the node's own broker credential from enrolment onward.
func freshPassword() (string, error) {
raw := make([]byte, 32)
if _, err := rand.Read(raw); err != nil {
return "", fmt.Errorf("cannot generate a broker password: %w", err)
}
return base64.RawURLEncoding.EncodeToString(raw), nil
}
var _ Enroller = Enrolment{}
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
var ErrNoBrokerManagement = errors.New("no broker management configured")
// Heard records what a node reported about itself.
//
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
// decides nothing from it.
func (e Enrolment) Heard(ctx context.Context, report Report) error {
if report.Node == "" {
return errors.New("a report named no node")
}
node, err := e.Inventory.NodeByName(ctx, report.Node)
if err != nil {
return err
}
// What it did is kept whichever way it went. Until this, a refusal or a failure moved
// last_seen and the reason went to a log line, so "which machine is not doing what it was
// told" had no answer the next morning — which is the question a mesh exists to answer.
doing := inventory.Doing{
Outcome: inventory.OutcomeApplied,
Refused: report.Refused,
Applied: len(report.Applied),
}
switch {
case report.Refused != "":
doing.Outcome = inventory.OutcomeRefused
case len(report.Failed) > 0:
doing.Outcome = inventory.OutcomeFailed
}
for id, why := range report.Failed {
doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why})
}
// Ordered, so two readings of one failure are the same reading.
sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID })
if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil {
return err
}
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
// as though it were the whole would tell a rebuilding node to remove what it still has.
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
return e.Inventory.Seen(ctx, node.ID)
}
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
}