Files
mesh-controller/internal/link/enrolment.go
T
jschoubben 646609c1b2 The mesh certifies names inside it
08-connectivity keeps two authorities apart on purpose: a public one for
names the outside world reaches, and the mesh's own for names only the
mesh knows. Nothing implemented the second, so anything between machines
was plaintext or trust-on-first-use — which the design refuses everywhere
else.

A node now generates a fourth key at enrolment and reports the public
half. A fourth, because a key used for two purposes is one rotation away
from breaking the other: the identity key signs messages to the mesh and
would do for TLS, and reusing it would mean rotating a node's identity
every time its certificate is replaced.

**Nothing secret travels and nothing is sealed.** A certificate authority
says "this name belongs to the holder of this key", so the mesh signs a
public half it cannot use, and the certificate it issues is public. A
module asks for one and is given the certificate and, if it wants,
the mesh's own — the private key is a path to a file the machine already
has, the same arrangement the private network's key uses.

Asserted by verifying rather than inspecting, because a certificate that
parses and does not chain fails at the moment something connects:

- what the mesh issues verifies against the mesh, for the name asked for
- the name is in the subject alternative names, since a certificate
  carrying it only in the common name is refused by every modern client
- it certifies the key the node generated and no other
- another mesh's certificate does not verify, which is the whole point of
  two authorities being separate
- the authority cannot sign another authority — one that could is one
  that can be delegated without anybody deciding to
- two control planes starting together agree on one authority, or a mesh
  has certificates half its machines refuse

Certificates last ten years, which is a choice: a short life needs
something to renew it, and a renewal that fails silently is a mesh that
stops trusting itself on a date nobody wrote down. What makes one
replaceable is that the mesh reissues on demand, not that it expires.
2026-08-31 00:09:13 +02:00

185 lines
7.2 KiB
Go

package link
import (
"context"
"crypto/ed25519"
"crypto/rand"
"encoding/base64"
"errors"
"fmt"
"sort"
"github.com/novox/mesh-control/internal/broker"
"github.com/novox/mesh-control/internal/identity"
"github.com/novox/mesh-control/internal/inventory"
)
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
// recorded, and the node gets its own queue.
//
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
// — it holds both grants and asks each for its part, which is what the process running them is
// for.
type Enrolment struct {
Inventory *inventory.Inventory
Identity *identity.Identity
Management *broker.Management
Broker broker.Broker
}
// Enrol spends the token and records what the node presented.
//
// Order matters and it is the order things become irreversible. The token is spent first, in a
// single statement that both finds and marks it, so two machines racing on one secret produce one
// winner. Only then is a key recorded — because recording a key for a node whose token turned out
// to be spent would leave the mesh believing a machine that never had the right to join.
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (EnrolReply, error) {
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
if len(public) != ed25519.PublicKeySize {
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
len(public), ed25519.PublicKeySize)
}
node, err := e.Inventory.Redeem(ctx, secret)
if err != nil {
return EnrolReply{}, err
}
// From here the token is gone whatever happens next, so anything that fails leaves a node
// record with no live key — which is visible and fixable with a new token, where a spent
// token believed to be unspent is neither.
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and the key could not be recorded, so %s has no identity and "+
"needs a new token: %w", node.Name, err)
}
key, err := e.Identity.Active(ctx)
if err != nil {
return EnrolReply{}, err
}
reply := EnrolReply{
Accepted: true,
Node: node.Name,
Queue: QueueFor(node.Name),
Broker: e.Broker.Address,
Fingerprint: e.Broker.Fingerprint,
Signer: key.Public,
}
// The token's secret was the broker password up to this moment, which is what let this
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
// credential the node keeps for years is not the one that was pasted into a terminal.
if e.Management != nil {
password, err := freshPassword()
if err != nil {
return EnrolReply{}, err
}
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's broker password could not be replaced: %w",
node.Name, err)
}
reply.Password = password
}
// Recorded before the profile because the overlay is the first declaration this node will
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
// briefly as possible.
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
// stronger: without it the mesh cannot send this node a credential at all, and a node that
// enrolled without one will be refused a sealed file rather than quietly given none.
// And the key it serves TLS with, so the mesh can certify its internal name. Public, so it is
// recorded rather than sealed — the node keeps the half that matters.
if request.ServingKey != "" {
if err := e.Identity.RecordServingKey(ctx, node.ID, request.ServingKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's serving key could not be recorded: %w",
node.Name, err)
}
}
if request.SealingKey != "" {
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's sealing key could not be recorded: %w",
node.Name, err)
}
}
if request.OverlayKey != "" {
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"the token was spent and %s's overlay key could not be recorded: %w", node.Name, err)
}
}
if profile != nil {
// Not fatal if it fails. The profile is what the control plane needs in order to decide
// what this machine should run, and it is reported again on every connection — so losing
// it here costs a decision that can be made later, not the enrolment.
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
}
return reply, nil
}
// freshPassword is the node's own broker credential from enrolment onward.
func freshPassword() (string, error) {
raw := make([]byte, 32)
if _, err := rand.Read(raw); err != nil {
return "", fmt.Errorf("cannot generate a broker password: %w", err)
}
return base64.RawURLEncoding.EncodeToString(raw), nil
}
var _ Enroller = Enrolment{}
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
var ErrNoBrokerManagement = errors.New("no broker management configured")
// Heard records what a node reported about itself.
//
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
// decides nothing from it.
func (e Enrolment) Heard(ctx context.Context, report Report) error {
if report.Node == "" {
return errors.New("a report named no node")
}
node, err := e.Inventory.NodeByName(ctx, report.Node)
if err != nil {
return err
}
// What it did is kept whichever way it went. Until this, a refusal or a failure moved
// last_seen and the reason went to a log line, so "which machine is not doing what it was
// told" had no answer the next morning — which is the question a mesh exists to answer.
doing := inventory.Doing{
Outcome: inventory.OutcomeApplied,
Refused: report.Refused,
Applied: len(report.Applied),
}
switch {
case report.Refused != "":
doing.Outcome = inventory.OutcomeRefused
case len(report.Failed) > 0:
doing.Outcome = inventory.OutcomeFailed
}
for id, why := range report.Failed {
doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why})
}
// Ordered, so two readings of one failure are the same reading.
sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID })
if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil {
return err
}
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
// as though it were the whole would tell a rebuilding node to remove what it still has.
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
return e.Inventory.Seen(ctx, node.ID)
}
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
}