On an adopted hub the private network takes over the tunnel it finds rather than running beside it (hq ADR 0105): two tunnels leave the mesh's unreachable through the provider's filter, so no machine can ever join. The node presents the found tunnel when it enrols, under the key it took as its own; the inventory records it (node.tunnel, tunnel_peer — migration 0031) and the mesh composes from it: the overlay's range is the adopted tunnel's, the hub is placed at the tunnel's address on the tunnel's port, and every peer the tunnel had is carried in the hub's peer list as a peer of the tunnel, not a node of the mesh, until a node enrols with that key — which then keeps the address the tunnel had for it. A fresh node never gets an address the tunnel holds. The hub's declaration tells the host which unit to take over; the host's account of carrying it is recorded and shown. Every reader of the range follows the setting; nothing stores it. A found tunnel under another key is recorded and not adopted, so ADR 0100's non-overlap rule keeps applying where a tunnel is left running beside the mesh's. A lab bed and test skeleton for "How it is checked" are under lab/.
308 lines
14 KiB
Go
308 lines
14 KiB
Go
package link
|
|
|
|
import (
|
|
"context"
|
|
"crypto/ed25519"
|
|
"crypto/rand"
|
|
"crypto/sha256"
|
|
"encoding/base64"
|
|
"encoding/hex"
|
|
"errors"
|
|
"fmt"
|
|
"log"
|
|
"sort"
|
|
|
|
"github.com/novox/mesh-controller/internal/broker"
|
|
"github.com/novox/mesh-controller/internal/identity"
|
|
"github.com/novox/mesh-controller/internal/inventory"
|
|
)
|
|
|
|
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
|
|
// recorded, and the node gets its own queue.
|
|
//
|
|
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
|
|
// — it holds both grants and asks each for its part, which is what the process running them is
|
|
// for.
|
|
type Enrolment struct {
|
|
Inventory *inventory.Inventory
|
|
Identity *identity.Identity
|
|
Management *broker.Management
|
|
Broker broker.Broker
|
|
}
|
|
|
|
// Enrol records what the node presented and spends the token.
|
|
//
|
|
// Order matters and it is the order things become irreversible (novox/hq issue 083). The token is
|
|
// claimed first, in a single statement that both finds it and holds it for this presenter's key, so
|
|
// two machines racing on one secret produce one holder. Then everything the node presented is
|
|
// written — each write an overwrite, so an attempt interrupted by the store going away can be made
|
|
// again by the same presenter. Then the token is spent. Last, the token's secret stops being the
|
|
// node's broker password: done after the spend, because a password replaced by an attempt that
|
|
// then failed would be one nobody holds, and the node could not even log in to ask again.
|
|
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (reply EnrolReply, err error) {
|
|
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
|
|
|
|
// A store that could not be asked right now, or a token another presenter holds for the
|
|
// moment, is "not now": the node asks again with the same request (novox/hq issue 083).
|
|
defer func() {
|
|
if inventory.Unreachable(err) || errors.Is(err, inventory.ErrTokenInUse) ||
|
|
errors.Is(err, context.Canceled) {
|
|
err = fmt.Errorf("%w: %w", ErrTryAgain, err)
|
|
}
|
|
}()
|
|
|
|
if len(public) != ed25519.PublicKeySize {
|
|
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
|
|
len(public), ed25519.PublicKeySize)
|
|
}
|
|
|
|
// Claimed, not spent: the token is held for this presenter while the node is written, and
|
|
// spent only as the last write. The node's identity lives in another database than the
|
|
// token, so the two cannot be one transaction; a failure between them used to leave a spent
|
|
// token and a node with no key, which the host — making new keys on every attempt — could not
|
|
// recover from. Every write below overwrites, so an attempt made again is safe.
|
|
// A proof that does not verify is refused outright: it was made with another key, or for
|
|
// another request. One that verifies lets this presenter finish an enrolment whose token it
|
|
// already spent — never a request the broker handed over a second time, which may already
|
|
// have been answered.
|
|
proven := false
|
|
if len(request.Proof) > 0 {
|
|
if !ed25519.Verify(public, EnrolProof(secret, public, request.OverlayKey, request.SealingKey,
|
|
request.ServingKey), request.Proof) {
|
|
return EnrolReply{}, errors.New("the enrolment's proof does not match the key it presents")
|
|
}
|
|
proven = true
|
|
}
|
|
by := claimant(public)
|
|
node, err := e.Inventory.Claim(ctx, secret, by, proven && !request.Redelivered)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
|
|
return EnrolReply{}, fmt.Errorf("%s's key could not be recorded: %w", node.Name, err)
|
|
}
|
|
|
|
key, err := e.Identity.Active(ctx)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
reply = EnrolReply{
|
|
Accepted: true,
|
|
Node: node.Name,
|
|
Queue: QueueFor(node.Name),
|
|
Broker: e.Broker.Address,
|
|
Fingerprint: e.Broker.Fingerprint,
|
|
Signer: key.Public,
|
|
}
|
|
|
|
// Recorded before the profile because the overlay is the first declaration this node will
|
|
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
|
|
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
|
|
// briefly as possible.
|
|
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
|
|
// stronger: without it the mesh cannot send this node a credential at all, and a node that
|
|
// enrolled without one will be refused a sealed file rather than quietly given none.
|
|
// And the key it serves TLS with, so the mesh can certify its internal name. Public, so it is
|
|
// recorded rather than sealed — the node keeps the half that matters.
|
|
if request.ServingKey != "" {
|
|
if err := e.Identity.RecordServingKey(ctx, node.ID, request.ServingKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"%s's serving key could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
if request.SealingKey != "" {
|
|
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"%s's sealing key could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
if request.OverlayKey != "" {
|
|
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"%s's overlay key could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
// And the tunnel it found, whose key is the overlay key above (novox/hq ADR 0105). Recorded
|
|
// before the token is spent for the same reason as the keys: the first declaration this node
|
|
// receives is composed from it, and a hub enrolled without its tunnel would be placed at an
|
|
// address of the mesh's choosing rather than the tunnel's.
|
|
if request.Tunnel != nil {
|
|
if request.Tunnel.PublicKey != request.OverlayKey {
|
|
return EnrolReply{}, fmt.Errorf("%s presented a tunnel under key %s and an overlay key "+
|
|
"that is not it; a tunnel is taken over with its own key or not at all", node.Name,
|
|
request.Tunnel.PublicKey)
|
|
}
|
|
peers := make([]inventory.TunnelPeer, 0, len(request.Tunnel.Peers))
|
|
for _, p := range request.Tunnel.Peers {
|
|
peers = append(peers, inventory.TunnelPeer{PublicKey: p.PublicKey, Address: p.Address})
|
|
}
|
|
if err := e.Inventory.RecordTunnel(ctx, node.ID, inventory.Tunnel{
|
|
Interface: request.Tunnel.Interface, Unit: request.Tunnel.Unit,
|
|
Config: request.Tunnel.Config, Port: request.Tunnel.Port,
|
|
Address: request.Tunnel.Address, Range: request.Tunnel.Range,
|
|
PublicKey: request.Tunnel.PublicKey, Peers: peers,
|
|
}); err != nil {
|
|
return EnrolReply{}, fmt.Errorf("%s's found tunnel could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
|
|
// Spent once the node is complete in the store.
|
|
if err := e.Inventory.Spend(ctx, secret, by); err != nil {
|
|
return EnrolReply{}, fmt.Errorf("%s was written and its token could not be spent: %w", node.Name, err)
|
|
}
|
|
|
|
// The token's secret was the broker password up to this moment, which is what let this
|
|
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
|
|
// credential the node keeps for years is not the one that was pasted into a terminal. After
|
|
// the spend and not before: a replaced password on an attempt that failed would be held by
|
|
// nobody. If the broker will not take it now, the enrolment still stands — the node keeps
|
|
// the token's secret as its password, which it is told, and which is said here.
|
|
if e.Management != nil {
|
|
password, err := freshPassword()
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
|
|
log.Printf("%s is enrolled and its broker password could not be replaced, so it keeps "+
|
|
"the token's secret as its password: %v", node.Name, err)
|
|
} else {
|
|
reply.Password = password
|
|
}
|
|
}
|
|
|
|
if profile != nil {
|
|
// Not fatal if it fails. The profile is what the control plane needs in order to decide
|
|
// what this machine should run, and it is reported again on every connection — so losing
|
|
// it here costs a decision that can be made later, not the enrolment.
|
|
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
|
|
}
|
|
return reply, nil
|
|
}
|
|
|
|
// claimant names the key presenting a token, so a claim can be held for it alone.
|
|
func claimant(public ed25519.PublicKey) string {
|
|
sum := sha256.Sum256(public)
|
|
return hex.EncodeToString(sum[:])
|
|
}
|
|
|
|
// freshPassword is the node's own broker credential from enrolment onward.
|
|
func freshPassword() (string, error) {
|
|
raw := make([]byte, 32)
|
|
if _, err := rand.Read(raw); err != nil {
|
|
return "", fmt.Errorf("cannot generate a broker password: %w", err)
|
|
}
|
|
return base64.RawURLEncoding.EncodeToString(raw), nil
|
|
}
|
|
|
|
var _ Enroller = Enrolment{}
|
|
|
|
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
|
|
var ErrNoBrokerManagement = errors.New("no broker management configured")
|
|
|
|
// Heard records what a node reported about itself.
|
|
//
|
|
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
|
|
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
|
|
// decides nothing from it.
|
|
func (e Enrolment) Heard(ctx context.Context, report Report) (err error) {
|
|
// A store that could not be asked right now is said as such, so the report is kept for
|
|
// another attempt rather than acknowledged and lost (novox/hq issue 082).
|
|
defer func() {
|
|
if inventory.Unreachable(err) {
|
|
err = fmt.Errorf("%w: %w", ErrTryAgain, err)
|
|
}
|
|
}()
|
|
if report.Node == "" {
|
|
return errors.New("a report named no node")
|
|
}
|
|
node, err := e.Inventory.NodeByName(ctx, report.Node)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
// What an adopted node holds, which firewall it found, and what is reachable on it (novox/hq
|
|
// ADR 0100). Recorded whenever a report carries any of it — a node reports these on its own
|
|
// schedule, when what it holds changes, not only after an apply — and never cleared by a
|
|
// report that carries none, which is every bare word that the node is there. An adopted node
|
|
// always names its firewall, so a report from one replaces all three, emptied held included.
|
|
if len(report.Held) > 0 || report.Firewall != "" || len(report.Reachable) > 0 {
|
|
held := make([]inventory.Held, 0, len(report.Held))
|
|
for _, h := range report.Held {
|
|
held = append(held, inventory.Held{ID: h.ID, Module: h.Module, Kind: h.Kind,
|
|
Target: h.Target, Since: h.Since, Changed: h.Changed, Kept: h.Kept})
|
|
}
|
|
reachable := make([]inventory.Reach, 0, len(report.Reachable))
|
|
for _, r := range report.Reachable {
|
|
reachable = append(reachable, inventory.Reach{Protocol: r.Protocol, Address: r.Address,
|
|
Port: r.Port, By: r.By, Published: r.Published, ContainerPort: r.ContainerPort})
|
|
}
|
|
if err := e.Inventory.RecordAdoption(ctx, node.ID, held, report.Firewall, reachable); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
// What it says about the tunnel it carried (novox/hq ADR 0105), whenever it says it.
|
|
if report.Tunnel != nil {
|
|
if err := e.Inventory.RecordCarriedTunnel(ctx, node.ID, inventory.Carried{
|
|
Interface: report.Tunnel.Interface, Port: report.Tunnel.Port, Range: report.Tunnel.Range,
|
|
Peers: report.Tunnel.Peers, Taken: report.Tunnel.Taken, Kept: report.Tunnel.Kept,
|
|
}); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
|
|
// A bare word that a node is there is not an account of what the machine did or holds: it
|
|
// moves last_seen and touches nothing else. This arrives every minute (link.AliveEvery),
|
|
// while a real report is rare, so recording it as one would overwrite the node's last real
|
|
// apply with an empty one — wiping the declaration digest that decides whether the node is
|
|
// current, the carried ports a push assigns around, and the clean-or-failed outcome — and a
|
|
// node that had just caught up would read as behind within the minute. The alive path calls
|
|
// this with only a node name; a real report always carries an account (something applied, or
|
|
// a refusal, or a failure), so those are the reports that get written down.
|
|
if report.Applied == nil && report.Refused == "" && len(report.Failed) == 0 {
|
|
if report.Superseded != "" {
|
|
log.Printf("%s set aside declaration %s for the newer %s", report.Node, report.Declared, report.Superseded)
|
|
}
|
|
return e.Inventory.Seen(ctx, node.ID)
|
|
}
|
|
// What it did is kept whichever way it went. Until this, a refusal or a failure moved
|
|
// last_seen and the reason went to a log line, so "which machine is not doing what it was
|
|
// told" had no answer the next morning — which is the question a mesh exists to answer.
|
|
doing := inventory.Doing{
|
|
Outcome: inventory.OutcomeApplied,
|
|
Refused: report.Refused,
|
|
Applied: len(report.Applied),
|
|
Declared: report.Declared,
|
|
}
|
|
switch {
|
|
case report.Refused != "":
|
|
doing.Outcome = inventory.OutcomeRefused
|
|
case len(report.Failed) > 0:
|
|
doing.Outcome = inventory.OutcomeFailed
|
|
}
|
|
for id, why := range report.Failed {
|
|
doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why})
|
|
}
|
|
// Ordered, so two readings of one failure are the same reading.
|
|
sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID })
|
|
|
|
// And what that machine says it already holds, so a port is assigned around it rather than
|
|
// on top of it (novox/hq ADR 0038). Kept even when the declaration was refused: what the
|
|
// machine carries is true regardless of what it thought of the last thing it was sent.
|
|
if err := e.Inventory.RecordCarried(ctx, report.Node, report.Carried); err != nil {
|
|
return err
|
|
}
|
|
if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil {
|
|
return err
|
|
}
|
|
|
|
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
|
|
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
|
|
// as though it were the whole would tell a rebuilding node to remove what it still has.
|
|
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
|
|
return e.Inventory.Seen(ctx, node.ID)
|
|
}
|
|
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
|
|
}
|