Files
mesh-controller/internal/link/enrolment.go
T
jschoubben 41f7c51032 Assign around what a machine already holds
The other half of ADR 0038, and what 04-ISSUES/028 was actually about.
A module can now avoid colliding with another module; until this it
could not avoid colliding with the mesh itself.

The substrate is not a module. A node raises it from the bundle it
carries before any mesh exists, so the control plane had never heard of
the store, the broker, or its own container — and handed a database
module 5432, which the store already had.

So the machine says. The host records what each resource binds,
distinguishing what it carried from what the mesh sent — a distinction
that already existed so the two never remove each other — and reports
the carried ones. The node states and this context writes, which is the
shape of every message between them.

What the declaration binds, not what is open. A machine's open ports are
a moving target, and assigning around them would mean a port that was
free when it was asked for and taken when it was used.

Replaced whole each time rather than merged: a machine that gave a port
back must be believed about that too, and a set that only grows keeps a
port reserved for something no longer there.

Tested against a real database, and the tests bite — removing the check
hands the module 20000, which the machine had said it holds.
2026-09-01 18:32:38 +02:00

192 lines
7.6 KiB
Go

package link
import (
"context"
"crypto/ed25519"
"crypto/rand"
"encoding/base64"
"errors"
"fmt"
"sort"
"github.com/novox/mesh-control/internal/broker"
"github.com/novox/mesh-control/internal/identity"
"github.com/novox/mesh-control/internal/inventory"
)
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
// recorded, and the node gets its own queue.
//
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
// — it holds both grants and asks each for its part, which is what the process running them is
// for.
type Enrolment struct {
Inventory *inventory.Inventory
Identity *identity.Identity
Management *broker.Management
Broker broker.Broker
}
// Enrol spends the token and records what the node presented.
//
// Order matters and it is the order things become irreversible. The token is spent first, in a
// single statement that both finds and marks it, so two machines racing on one secret produce one
// winner. Only then is a key recorded — because recording a key for a node whose token turned out
// to be spent would leave the mesh believing a machine that never had the right to join.
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (EnrolReply, error) {
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
if len(public) != ed25519.PublicKeySize {
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
len(public), ed25519.PublicKeySize)
}
node, err := e.Inventory.Redeem(ctx, secret)
if err != nil {
return EnrolReply{}, err
}
// From here the token is gone whatever happens next, so anything that fails leaves a node
// record with no live key — which is visible and fixable with a new token, where a spent
// token believed to be unspent is neither.
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and the key could not be recorded, so %s has no identity and "+
"needs a new token: %w", node.Name, err)
}
key, err := e.Identity.Active(ctx)
if err != nil {
return EnrolReply{}, err
}
reply := EnrolReply{
Accepted: true,
Node: node.Name,
Queue: QueueFor(node.Name),
Broker: e.Broker.Address,
Fingerprint: e.Broker.Fingerprint,
Signer: key.Public,
}
// The token's secret was the broker password up to this moment, which is what let this
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
// credential the node keeps for years is not the one that was pasted into a terminal.
if e.Management != nil {
password, err := freshPassword()
if err != nil {
return EnrolReply{}, err
}
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's broker password could not be replaced: %w",
node.Name, err)
}
reply.Password = password
}
// Recorded before the profile because the overlay is the first declaration this node will
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
// briefly as possible.
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
// stronger: without it the mesh cannot send this node a credential at all, and a node that
// enrolled without one will be refused a sealed file rather than quietly given none.
// And the key it serves TLS with, so the mesh can certify its internal name. Public, so it is
// recorded rather than sealed — the node keeps the half that matters.
if request.ServingKey != "" {
if err := e.Identity.RecordServingKey(ctx, node.ID, request.ServingKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's serving key could not be recorded: %w",
node.Name, err)
}
}
if request.SealingKey != "" {
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's sealing key could not be recorded: %w",
node.Name, err)
}
}
if request.OverlayKey != "" {
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's overlay key could not be recorded: %w", node.Name, err)
}
}
if profile != nil {
// Not fatal if it fails. The profile is what the control plane needs in order to decide
// what this machine should run, and it is reported again on every connection — so losing
// it here costs a decision that can be made later, not the enrolment.
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
}
return reply, nil
}
// freshPassword is the node's own broker credential from enrolment onward.
func freshPassword() (string, error) {
raw := make([]byte, 32)
if _, err := rand.Read(raw); err != nil {
return "", fmt.Errorf("cannot generate a broker password: %w", err)
}
return base64.RawURLEncoding.EncodeToString(raw), nil
}
var _ Enroller = Enrolment{}
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
var ErrNoBrokerManagement = errors.New("no broker management configured")
// Heard records what a node reported about itself.
//
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
// decides nothing from it.
func (e Enrolment) Heard(ctx context.Context, report Report) error {
if report.Node == "" {
return errors.New("a report named no node")
}
node, err := e.Inventory.NodeByName(ctx, report.Node)
if err != nil {
return err
}
// What it did is kept whichever way it went. Until this, a refusal or a failure moved
// last_seen and the reason went to a log line, so "which machine is not doing what it was
// told" had no answer the next morning — which is the question a mesh exists to answer.
doing := inventory.Doing{
Outcome: inventory.OutcomeApplied,
Refused: report.Refused,
Applied: len(report.Applied),
}
switch {
case report.Refused != "":
doing.Outcome = inventory.OutcomeRefused
case len(report.Failed) > 0:
doing.Outcome = inventory.OutcomeFailed
}
for id, why := range report.Failed {
doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why})
}
// Ordered, so two readings of one failure are the same reading.
sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID })
// And what that machine says it already holds, so a port is assigned around it rather than
// on top of it (novox/hq ADR 0038). Kept even when the declaration was refused: what the
// machine carries is true regardless of what it thought of the last thing it was sent.
if err := e.Inventory.RecordCarried(ctx, report.Node, report.Carried); err != nil {
return err
}
if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil {
return err
}
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
// as though it were the whole would tell a rebuilding node to remove what it still has.
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
return e.Inventory.Seen(ctx, node.ID)
}
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
}