The other half of ADR 0038, and what 04-ISSUES/028 was actually about. A module can now avoid colliding with another module; until this it could not avoid colliding with the mesh itself. The substrate is not a module. A node raises it from the bundle it carries before any mesh exists, so the control plane had never heard of the store, the broker, or its own container — and handed a database module 5432, which the store already had. So the machine says. The host records what each resource binds, distinguishing what it carried from what the mesh sent — a distinction that already existed so the two never remove each other — and reports the carried ones. The node states and this context writes, which is the shape of every message between them. What the declaration binds, not what is open. A machine's open ports are a moving target, and assigning around them would mean a port that was free when it was asked for and taken when it was used. Replaced whole each time rather than merged: a machine that gave a port back must be believed about that too, and a set that only grows keeps a port reserved for something no longer there. Tested against a real database, and the tests bite — removing the check hands the module 20000, which the machine had said it holds.
192 lines
7.6 KiB
Go
192 lines
7.6 KiB
Go
package link
|
|
|
|
import (
|
|
"context"
|
|
"crypto/ed25519"
|
|
"crypto/rand"
|
|
"encoding/base64"
|
|
"errors"
|
|
"fmt"
|
|
"sort"
|
|
|
|
"github.com/novox/mesh-control/internal/broker"
|
|
"github.com/novox/mesh-control/internal/identity"
|
|
"github.com/novox/mesh-control/internal/inventory"
|
|
)
|
|
|
|
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
|
|
// recorded, and the node gets its own queue.
|
|
//
|
|
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
|
|
// — it holds both grants and asks each for its part, which is what the process running them is
|
|
// for.
|
|
type Enrolment struct {
|
|
Inventory *inventory.Inventory
|
|
Identity *identity.Identity
|
|
Management *broker.Management
|
|
Broker broker.Broker
|
|
}
|
|
|
|
// Enrol spends the token and records what the node presented.
|
|
//
|
|
// Order matters and it is the order things become irreversible. The token is spent first, in a
|
|
// single statement that both finds and marks it, so two machines racing on one secret produce one
|
|
// winner. Only then is a key recorded — because recording a key for a node whose token turned out
|
|
// to be spent would leave the mesh believing a machine that never had the right to join.
|
|
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (EnrolReply, error) {
|
|
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
|
|
|
|
if len(public) != ed25519.PublicKeySize {
|
|
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
|
|
len(public), ed25519.PublicKeySize)
|
|
}
|
|
|
|
node, err := e.Inventory.Redeem(ctx, secret)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
// From here the token is gone whatever happens next, so anything that fails leaves a node
|
|
// record with no live key — which is visible and fixable with a new token, where a spent
|
|
// token believed to be unspent is neither.
|
|
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and the key could not be recorded, so %s has no identity and "+
|
|
"needs a new token: %w", node.Name, err)
|
|
}
|
|
|
|
key, err := e.Identity.Active(ctx)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
reply := EnrolReply{
|
|
Accepted: true,
|
|
Node: node.Name,
|
|
Queue: QueueFor(node.Name),
|
|
Broker: e.Broker.Address,
|
|
Fingerprint: e.Broker.Fingerprint,
|
|
Signer: key.Public,
|
|
}
|
|
|
|
// The token's secret was the broker password up to this moment, which is what let this
|
|
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
|
|
// credential the node keeps for years is not the one that was pasted into a terminal.
|
|
if e.Management != nil {
|
|
password, err := freshPassword()
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's broker password could not be replaced: %w",
|
|
node.Name, err)
|
|
}
|
|
reply.Password = password
|
|
}
|
|
|
|
// Recorded before the profile because the overlay is the first declaration this node will
|
|
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
|
|
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
|
|
// briefly as possible.
|
|
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
|
|
// stronger: without it the mesh cannot send this node a credential at all, and a node that
|
|
// enrolled without one will be refused a sealed file rather than quietly given none.
|
|
// And the key it serves TLS with, so the mesh can certify its internal name. Public, so it is
|
|
// recorded rather than sealed — the node keeps the half that matters.
|
|
if request.ServingKey != "" {
|
|
if err := e.Identity.RecordServingKey(ctx, node.ID, request.ServingKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's serving key could not be recorded: %w",
|
|
node.Name, err)
|
|
}
|
|
}
|
|
if request.SealingKey != "" {
|
|
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's sealing key could not be recorded: %w",
|
|
node.Name, err)
|
|
}
|
|
}
|
|
if request.OverlayKey != "" {
|
|
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's overlay key could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
|
|
if profile != nil {
|
|
// Not fatal if it fails. The profile is what the control plane needs in order to decide
|
|
// what this machine should run, and it is reported again on every connection — so losing
|
|
// it here costs a decision that can be made later, not the enrolment.
|
|
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
|
|
}
|
|
return reply, nil
|
|
}
|
|
|
|
// freshPassword is the node's own broker credential from enrolment onward.
|
|
func freshPassword() (string, error) {
|
|
raw := make([]byte, 32)
|
|
if _, err := rand.Read(raw); err != nil {
|
|
return "", fmt.Errorf("cannot generate a broker password: %w", err)
|
|
}
|
|
return base64.RawURLEncoding.EncodeToString(raw), nil
|
|
}
|
|
|
|
var _ Enroller = Enrolment{}
|
|
|
|
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
|
|
var ErrNoBrokerManagement = errors.New("no broker management configured")
|
|
|
|
// Heard records what a node reported about itself.
|
|
//
|
|
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
|
|
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
|
|
// decides nothing from it.
|
|
func (e Enrolment) Heard(ctx context.Context, report Report) error {
|
|
if report.Node == "" {
|
|
return errors.New("a report named no node")
|
|
}
|
|
node, err := e.Inventory.NodeByName(ctx, report.Node)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
// What it did is kept whichever way it went. Until this, a refusal or a failure moved
|
|
// last_seen and the reason went to a log line, so "which machine is not doing what it was
|
|
// told" had no answer the next morning — which is the question a mesh exists to answer.
|
|
doing := inventory.Doing{
|
|
Outcome: inventory.OutcomeApplied,
|
|
Refused: report.Refused,
|
|
Applied: len(report.Applied),
|
|
}
|
|
switch {
|
|
case report.Refused != "":
|
|
doing.Outcome = inventory.OutcomeRefused
|
|
case len(report.Failed) > 0:
|
|
doing.Outcome = inventory.OutcomeFailed
|
|
}
|
|
for id, why := range report.Failed {
|
|
doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why})
|
|
}
|
|
// Ordered, so two readings of one failure are the same reading.
|
|
sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID })
|
|
|
|
// And what that machine says it already holds, so a port is assigned around it rather than
|
|
// on top of it (novox/hq ADR 0038). Kept even when the declaration was refused: what the
|
|
// machine carries is true regardless of what it thought of the last thing it was sent.
|
|
if err := e.Inventory.RecordCarried(ctx, report.Node, report.Carried); err != nil {
|
|
return err
|
|
}
|
|
if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil {
|
|
return err
|
|
}
|
|
|
|
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
|
|
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
|
|
// as though it were the whole would tell a rebuilding node to remove what it still has.
|
|
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
|
|
return e.Inventory.Seen(ctx, node.ID)
|
|
}
|
|
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
|
|
}
|