Three machines across two sites, two of them behind no reachable address, all nine paths open. The mesh computes the graph, delivers it as a declaration, and the nodes bring it up. Every fault below looked like success from inside the mesh: the graph was right, the files were right, the services were up, every node reported it had applied. None was reachable by reasoning. A running interface does not re-read its configuration. A node joins, every existing node's peer list changes, the file is replaced -- and the service is already running, so nothing reloads it. Fixed as declared state rather than a command: the service must reflect the file. A command to restart would be an action, and the link may not carry one. The host refused exactly that, which is how this shape was arrived at. A hub sharing a site with a spoke appeared twice in that spoke's peer list -- once as a direct peer, once as the route of last resort. WireGuard takes one entry per key and refuses the file. The ordinary shape of a small mesh, and in none of the tests written before it ran. Two nodes at one site that neither can be dialled were peered directly. Nobody opens the path, and the direct route is more specific than the hub's, so it wins and blackholes -- this design's own warning arriving in its implementation. They now route through the hub unless one end can be dialled. And Docker sets the FORWARD policy to DROP, so a hub with ip_forward enabled carried nothing between its spokes. The substrate at tier 1 silently breaks the network at tier 2, and nothing in either tier's state says so. The hub inserts its own rule above those chains and removes it on the way down. Two weak tests found by injection along the way: one asserted the keepalive rule only against the hub, whose peer entries happen not to set that field at all, so it tested an absence; the other checked the firewall rules by looking for FORWARD anywhere, which the PostDown line satisfies on its own.
142 lines
5.3 KiB
Go
142 lines
5.3 KiB
Go
package link
|
|
|
|
import (
|
|
"context"
|
|
"crypto/ed25519"
|
|
"crypto/rand"
|
|
"encoding/base64"
|
|
"errors"
|
|
"fmt"
|
|
|
|
"github.com/novox/mesh-control/internal/broker"
|
|
"github.com/novox/mesh-control/internal/identity"
|
|
"github.com/novox/mesh-control/internal/inventory"
|
|
)
|
|
|
|
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
|
|
// recorded, and the node gets its own queue.
|
|
//
|
|
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
|
|
// — it holds both grants and asks each for its part, which is what the process running them is
|
|
// for.
|
|
type Enrolment struct {
|
|
Inventory *inventory.Inventory
|
|
Identity *identity.Identity
|
|
Management *broker.Management
|
|
Broker broker.Broker
|
|
}
|
|
|
|
// Enrol spends the token and records what the node presented.
|
|
//
|
|
// Order matters and it is the order things become irreversible. The token is spent first, in a
|
|
// single statement that both finds and marks it, so two machines racing on one secret produce one
|
|
// winner. Only then is a key recorded — because recording a key for a node whose token turned out
|
|
// to be spent would leave the mesh believing a machine that never had the right to join.
|
|
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (EnrolReply, error) {
|
|
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
|
|
|
|
if len(public) != ed25519.PublicKeySize {
|
|
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
|
|
len(public), ed25519.PublicKeySize)
|
|
}
|
|
|
|
node, err := e.Inventory.Redeem(ctx, secret)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
// From here the token is gone whatever happens next, so anything that fails leaves a node
|
|
// record with no live key — which is visible and fixable with a new token, where a spent
|
|
// token believed to be unspent is neither.
|
|
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and the key could not be recorded, so %s has no identity and "+
|
|
"needs a new token: %w", node.Name, err)
|
|
}
|
|
|
|
key, err := e.Identity.Active(ctx)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
reply := EnrolReply{
|
|
Accepted: true,
|
|
Node: node.Name,
|
|
Queue: QueueFor(node.Name),
|
|
Broker: e.Broker.Address,
|
|
Fingerprint: e.Broker.Fingerprint,
|
|
Signer: key.Public,
|
|
}
|
|
|
|
// The token's secret was the broker password up to this moment, which is what let this
|
|
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
|
|
// credential the node keeps for years is not the one that was pasted into a terminal.
|
|
if e.Management != nil {
|
|
password, err := freshPassword()
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's broker password could not be replaced: %w",
|
|
node.Name, err)
|
|
}
|
|
reply.Password = password
|
|
}
|
|
|
|
// Recorded before the profile because the overlay is the first declaration this node will
|
|
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
|
|
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
|
|
// briefly as possible.
|
|
if request.OverlayKey != "" {
|
|
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's overlay key could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
|
|
if profile != nil {
|
|
// Not fatal if it fails. The profile is what the control plane needs in order to decide
|
|
// what this machine should run, and it is reported again on every connection — so losing
|
|
// it here costs a decision that can be made later, not the enrolment.
|
|
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
|
|
}
|
|
return reply, nil
|
|
}
|
|
|
|
// freshPassword is the node's own broker credential from enrolment onward.
|
|
func freshPassword() (string, error) {
|
|
raw := make([]byte, 32)
|
|
if _, err := rand.Read(raw); err != nil {
|
|
return "", fmt.Errorf("cannot generate a broker password: %w", err)
|
|
}
|
|
return base64.RawURLEncoding.EncodeToString(raw), nil
|
|
}
|
|
|
|
var _ Enroller = Enrolment{}
|
|
|
|
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
|
|
var ErrNoBrokerManagement = errors.New("no broker management configured")
|
|
|
|
// Heard records what a node reported about itself.
|
|
//
|
|
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
|
|
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
|
|
// decides nothing from it.
|
|
func (e Enrolment) Heard(ctx context.Context, report Report) error {
|
|
if report.Node == "" {
|
|
return errors.New("a report named no node")
|
|
}
|
|
node, err := e.Inventory.NodeByName(ctx, report.Node)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
// A refusal or a failure is not an account of what the machine holds, so it moves last_seen
|
|
// and nothing else. Recording a partial list as though it were the whole would tell a
|
|
// rebuilding node to remove what it still has.
|
|
if report.Refused != "" || len(report.Failed) > 0 {
|
|
return e.Inventory.Seen(ctx, node.ID)
|
|
}
|
|
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
|
|
}
|