Files
mesh-controller/internal/link/enrolment.go
T

254 lines
11 KiB
Go

package link
import (
"context"
"crypto/ed25519"
"crypto/rand"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
"errors"
"fmt"
"log"
"sort"
"github.com/novox/mesh-controller/internal/broker"
"github.com/novox/mesh-controller/internal/identity"
"github.com/novox/mesh-controller/internal/inventory"
)
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
// recorded, and the node gets its own queue.
//
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
// — it holds both grants and asks each for its part, which is what the process running them is
// for.
type Enrolment struct {
Inventory *inventory.Inventory
Identity *identity.Identity
Management *broker.Management
Broker broker.Broker
}
// Enrol records what the node presented and spends the token.
//
// Order matters and it is the order things become irreversible (novox/hq issue 083). The token is
// claimed first, in a single statement that both finds it and holds it for this presenter's key, so
// two machines racing on one secret produce one holder. Then everything the node presented is
// written — each write an overwrite, so an attempt interrupted by the store going away can be made
// again by the same presenter. Then the token is spent. Last, the token's secret stops being the
// node's broker password: done after the spend, because a password replaced by an attempt that
// then failed would be one nobody holds, and the node could not even log in to ask again.
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (reply EnrolReply, err error) {
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
// A store that could not be asked right now, or a token another presenter holds for the
// moment, is "not now": the node asks again with the same request (novox/hq issue 083).
defer func() {
if inventory.Unreachable(err) || errors.Is(err, inventory.ErrTokenInUse) {
err = fmt.Errorf("%w: %w", ErrTryAgain, err)
}
}()
if len(public) != ed25519.PublicKeySize {
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
len(public), ed25519.PublicKeySize)
}
// Claimed, not spent: the token is held for this presenter while the node is written, and
// spent only as the last write. The node's identity lives in another database than the
// token, so the two cannot be one transaction; a failure between them used to leave a spent
// token and a node with no key, which the host — making new keys on every attempt — could not
// recover from. Every write below overwrites, so an attempt made again is safe.
// A proof that does not verify is refused outright: it was made with another key, or for
// another request. One that verifies lets this presenter finish an enrolment whose token it
// already spent — never a request the broker handed over a second time, which may already
// have been answered.
proven := false
if len(request.Proof) > 0 {
if !ed25519.Verify(public, EnrolProof(secret, public, request.OverlayKey, request.SealingKey,
request.ServingKey), request.Proof) {
return EnrolReply{}, errors.New("the enrolment's proof does not match the key it presents")
}
proven = true
}
by := claimant(public)
node, err := e.Inventory.Claim(ctx, secret, by, proven && !request.Redelivered)
if err != nil {
return EnrolReply{}, err
}
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
return EnrolReply{}, fmt.Errorf("%s's key could not be recorded: %w", node.Name, err)
}
key, err := e.Identity.Active(ctx)
if err != nil {
return EnrolReply{}, err
}
reply = EnrolReply{
Accepted: true,
Node: node.Name,
Queue: QueueFor(node.Name),
Broker: e.Broker.Address,
Fingerprint: e.Broker.Fingerprint,
Signer: key.Public,
}
// Recorded before the profile because the overlay is the first declaration this node will
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
// briefly as possible.
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
// stronger: without it the mesh cannot send this node a credential at all, and a node that
// enrolled without one will be refused a sealed file rather than quietly given none.
// And the key it serves TLS with, so the mesh can certify its internal name. Public, so it is
// recorded rather than sealed — the node keeps the half that matters.
if request.ServingKey != "" {
if err := e.Identity.RecordServingKey(ctx, node.ID, request.ServingKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"%s's serving key could not be recorded: %w", node.Name, err)
}
}
if request.SealingKey != "" {
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"%s's sealing key could not be recorded: %w", node.Name, err)
}
}
if request.OverlayKey != "" {
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
return EnrolReply{}, fmt.Errorf(
"%s's overlay key could not be recorded: %w", node.Name, err)
}
}
// Spent once the node is complete in the store.
if err := e.Inventory.Spend(ctx, secret, by); err != nil {
return EnrolReply{}, fmt.Errorf("%s was written and its token could not be spent: %w", node.Name, err)
}
// The token's secret was the broker password up to this moment, which is what let this
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
// credential the node keeps for years is not the one that was pasted into a terminal. After
// the spend and not before: a replaced password on an attempt that failed would be held by
// nobody. If the broker will not take it now, the enrolment still stands — the node keeps
// the token's secret as its password, which it is told, and which is said here.
if e.Management != nil {
password, err := freshPassword()
if err != nil {
return EnrolReply{}, err
}
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
log.Printf("%s is enrolled and its broker password could not be replaced, so it keeps "+
"the token's secret as its password: %v", node.Name, err)
} else {
reply.Password = password
}
}
if profile != nil {
// Not fatal if it fails. The profile is what the control plane needs in order to decide
// what this machine should run, and it is reported again on every connection — so losing
// it here costs a decision that can be made later, not the enrolment.
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
}
return reply, nil
}
// claimant names the key presenting a token, so a claim can be held for it alone.
func claimant(public ed25519.PublicKey) string {
sum := sha256.Sum256(public)
return hex.EncodeToString(sum[:])
}
// freshPassword is the node's own broker credential from enrolment onward.
func freshPassword() (string, error) {
raw := make([]byte, 32)
if _, err := rand.Read(raw); err != nil {
return "", fmt.Errorf("cannot generate a broker password: %w", err)
}
return base64.RawURLEncoding.EncodeToString(raw), nil
}
var _ Enroller = Enrolment{}
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
var ErrNoBrokerManagement = errors.New("no broker management configured")
// Heard records what a node reported about itself.
//
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
// decides nothing from it.
func (e Enrolment) Heard(ctx context.Context, report Report) (err error) {
// A store that could not be asked right now is said as such, so the report is kept for
// another attempt rather than acknowledged and lost (novox/hq issue 082).
defer func() {
if inventory.Unreachable(err) {
err = fmt.Errorf("%w: %w", ErrTryAgain, err)
}
}()
if report.Node == "" {
return errors.New("a report named no node")
}
node, err := e.Inventory.NodeByName(ctx, report.Node)
if err != nil {
return err
}
// A bare word that a node is there is not an account of what the machine did or holds: it
// moves last_seen and touches nothing else. This arrives every minute (link.AliveEvery),
// while a real report is rare, so recording it as one would overwrite the node's last real
// apply with an empty one — wiping the declaration digest that decides whether the node is
// current, the carried ports a push assigns around, and the clean-or-failed outcome — and a
// node that had just caught up would read as behind within the minute. The alive path calls
// this with only a node name; a real report always carries an account (something applied, or
// a refusal, or a failure), so those are the reports that get written down.
if report.Applied == nil && report.Refused == "" && len(report.Failed) == 0 {
if report.Superseded != "" {
log.Printf("%s set aside declaration %s for the newer %s", report.Node, report.Declared, report.Superseded)
}
return e.Inventory.Seen(ctx, node.ID)
}
// What it did is kept whichever way it went. Until this, a refusal or a failure moved
// last_seen and the reason went to a log line, so "which machine is not doing what it was
// told" had no answer the next morning — which is the question a mesh exists to answer.
doing := inventory.Doing{
Outcome: inventory.OutcomeApplied,
Refused: report.Refused,
Applied: len(report.Applied),
Declared: report.Declared,
}
switch {
case report.Refused != "":
doing.Outcome = inventory.OutcomeRefused
case len(report.Failed) > 0:
doing.Outcome = inventory.OutcomeFailed
}
for id, why := range report.Failed {
doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why})
}
// Ordered, so two readings of one failure are the same reading.
sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID })
// And what that machine says it already holds, so a port is assigned around it rather than
// on top of it (novox/hq ADR 0038). Kept even when the declaration was refused: what the
// machine carries is true regardless of what it thought of the last thing it was sent.
if err := e.Inventory.RecordCarried(ctx, report.Node, report.Carried); err != nil {
return err
}
if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil {
return err
}
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
// as though it were the whole would tell a rebuilding node to remove what it still has.
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
return e.Inventory.Seen(ctx, node.ID)
}
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
}