08-connectivity keeps two authorities apart on purpose: a public one for names the outside world reaches, and the mesh's own for names only the mesh knows. Nothing implemented the second, so anything between machines was plaintext or trust-on-first-use — which the design refuses everywhere else. A node now generates a fourth key at enrolment and reports the public half. A fourth, because a key used for two purposes is one rotation away from breaking the other: the identity key signs messages to the mesh and would do for TLS, and reusing it would mean rotating a node's identity every time its certificate is replaced. **Nothing secret travels and nothing is sealed.** A certificate authority says "this name belongs to the holder of this key", so the mesh signs a public half it cannot use, and the certificate it issues is public. A module asks for one and is given the certificate and, if it wants, the mesh's own — the private key is a path to a file the machine already has, the same arrangement the private network's key uses. Asserted by verifying rather than inspecting, because a certificate that parses and does not chain fails at the moment something connects: - what the mesh issues verifies against the mesh, for the name asked for - the name is in the subject alternative names, since a certificate carrying it only in the common name is refused by every modern client - it certifies the key the node generated and no other - another mesh's certificate does not verify, which is the whole point of two authorities being separate - the authority cannot sign another authority — one that could is one that can be delegated without anybody deciding to - two control planes starting together agree on one authority, or a mesh has certificates half its machines refuse Certificates last ten years, which is a choice: a short life needs something to renew it, and a renewal that fails silently is a mesh that stops trusting itself on a date nobody wrote down. What makes one replaceable is that the mesh reissues on demand, not that it expires.
185 lines
7.2 KiB
Go
185 lines
7.2 KiB
Go
package link
|
|
|
|
import (
|
|
"context"
|
|
"crypto/ed25519"
|
|
"crypto/rand"
|
|
"encoding/base64"
|
|
"errors"
|
|
"fmt"
|
|
"sort"
|
|
|
|
"github.com/novox/mesh-control/internal/broker"
|
|
"github.com/novox/mesh-control/internal/identity"
|
|
"github.com/novox/mesh-control/internal/inventory"
|
|
)
|
|
|
|
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
|
|
// recorded, and the node gets its own queue.
|
|
//
|
|
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
|
|
// — it holds both grants and asks each for its part, which is what the process running them is
|
|
// for.
|
|
type Enrolment struct {
|
|
Inventory *inventory.Inventory
|
|
Identity *identity.Identity
|
|
Management *broker.Management
|
|
Broker broker.Broker
|
|
}
|
|
|
|
// Enrol spends the token and records what the node presented.
|
|
//
|
|
// Order matters and it is the order things become irreversible. The token is spent first, in a
|
|
// single statement that both finds and marks it, so two machines racing on one secret produce one
|
|
// winner. Only then is a key recorded — because recording a key for a node whose token turned out
|
|
// to be spent would leave the mesh believing a machine that never had the right to join.
|
|
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (EnrolReply, error) {
|
|
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
|
|
|
|
if len(public) != ed25519.PublicKeySize {
|
|
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
|
|
len(public), ed25519.PublicKeySize)
|
|
}
|
|
|
|
node, err := e.Inventory.Redeem(ctx, secret)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
// From here the token is gone whatever happens next, so anything that fails leaves a node
|
|
// record with no live key — which is visible and fixable with a new token, where a spent
|
|
// token believed to be unspent is neither.
|
|
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and the key could not be recorded, so %s has no identity and "+
|
|
"needs a new token: %w", node.Name, err)
|
|
}
|
|
|
|
key, err := e.Identity.Active(ctx)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
reply := EnrolReply{
|
|
Accepted: true,
|
|
Node: node.Name,
|
|
Queue: QueueFor(node.Name),
|
|
Broker: e.Broker.Address,
|
|
Fingerprint: e.Broker.Fingerprint,
|
|
Signer: key.Public,
|
|
}
|
|
|
|
// The token's secret was the broker password up to this moment, which is what let this
|
|
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
|
|
// credential the node keeps for years is not the one that was pasted into a terminal.
|
|
if e.Management != nil {
|
|
password, err := freshPassword()
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's broker password could not be replaced: %w",
|
|
node.Name, err)
|
|
}
|
|
reply.Password = password
|
|
}
|
|
|
|
// Recorded before the profile because the overlay is the first declaration this node will
|
|
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
|
|
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
|
|
// briefly as possible.
|
|
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
|
|
// stronger: without it the mesh cannot send this node a credential at all, and a node that
|
|
// enrolled without one will be refused a sealed file rather than quietly given none.
|
|
// And the key it serves TLS with, so the mesh can certify its internal name. Public, so it is
|
|
// recorded rather than sealed — the node keeps the half that matters.
|
|
if request.ServingKey != "" {
|
|
if err := e.Identity.RecordServingKey(ctx, node.ID, request.ServingKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's serving key could not be recorded: %w",
|
|
node.Name, err)
|
|
}
|
|
}
|
|
if request.SealingKey != "" {
|
|
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's sealing key could not be recorded: %w",
|
|
node.Name, err)
|
|
}
|
|
}
|
|
if request.OverlayKey != "" {
|
|
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's overlay key could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
|
|
if profile != nil {
|
|
// Not fatal if it fails. The profile is what the control plane needs in order to decide
|
|
// what this machine should run, and it is reported again on every connection — so losing
|
|
// it here costs a decision that can be made later, not the enrolment.
|
|
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
|
|
}
|
|
return reply, nil
|
|
}
|
|
|
|
// freshPassword is the node's own broker credential from enrolment onward.
|
|
func freshPassword() (string, error) {
|
|
raw := make([]byte, 32)
|
|
if _, err := rand.Read(raw); err != nil {
|
|
return "", fmt.Errorf("cannot generate a broker password: %w", err)
|
|
}
|
|
return base64.RawURLEncoding.EncodeToString(raw), nil
|
|
}
|
|
|
|
var _ Enroller = Enrolment{}
|
|
|
|
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
|
|
var ErrNoBrokerManagement = errors.New("no broker management configured")
|
|
|
|
// Heard records what a node reported about itself.
|
|
//
|
|
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
|
|
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
|
|
// decides nothing from it.
|
|
func (e Enrolment) Heard(ctx context.Context, report Report) error {
|
|
if report.Node == "" {
|
|
return errors.New("a report named no node")
|
|
}
|
|
node, err := e.Inventory.NodeByName(ctx, report.Node)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
// What it did is kept whichever way it went. Until this, a refusal or a failure moved
|
|
// last_seen and the reason went to a log line, so "which machine is not doing what it was
|
|
// told" had no answer the next morning — which is the question a mesh exists to answer.
|
|
doing := inventory.Doing{
|
|
Outcome: inventory.OutcomeApplied,
|
|
Refused: report.Refused,
|
|
Applied: len(report.Applied),
|
|
}
|
|
switch {
|
|
case report.Refused != "":
|
|
doing.Outcome = inventory.OutcomeRefused
|
|
case len(report.Failed) > 0:
|
|
doing.Outcome = inventory.OutcomeFailed
|
|
}
|
|
for id, why := range report.Failed {
|
|
doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why})
|
|
}
|
|
// Ordered, so two readings of one failure are the same reading.
|
|
sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID })
|
|
if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil {
|
|
return err
|
|
}
|
|
|
|
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
|
|
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
|
|
// as though it were the whole would tell a rebuilding node to remove what it still has.
|
|
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
|
|
return e.Inventory.Seen(ctx, node.ID)
|
|
}
|
|
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
|
|
}
|