HAL keeps env vars in the registry, encrypted at rest. Its own tooling records what that bought and what it did not. `secret_locate` matches by value rather than by name — because the same password sits in mesh_provisions, in module_env, in each node's .env in plain text, and inside every connection string composed from it, and its documentation says those URL copies "are often the only copies actually in use". And a query against the encrypted column returns zero rows and proves nothing, so auditing moved to the decrypted copies on the nodes. Two faults there, and encryption at rest addresses neither: the control plane can read what it stores, so a copy of the database is a copy of every credential; and one secret has many homes with nothing tracking them. So here the mesh generates a password, seals it to each end with keys those nodes generated, stores both blobs, and discards the plaintext. It cannot read what it holds. Neither can the broker relaying it. And nothing is composed centrally — a connection string is assembled on the machine that needs one — so no copy is ever minted in a shape nothing tracks. `Compromise of a node is compromise of that node` (ADR 0004) is now true of secrets, not only of identity. Two files rather than one, because the mesh cannot compose a document containing a value it discarded: `binds` carries the readable facts, `secrets` carries the credential alone. The readable half stays readable in the declaration; the secret half changes only when the secret does, which makes restart-on precise. The provider gets a directory, one file per consumer, for the same reason. It is made once and kept — regenerating per declaration would restart both ends on every push, and the password a provider was told to create would never be the one its consumer was given. It is remade when either end's sealing key changes, and both ends learn the new one in the same push, so there is no window where half the mesh holds a dead credential. Two tests found passing for the wrong reason, both caught because their injection came back clean: - the provider's copy was asserted non-empty, which reads the same whichever column is selected. It now opens the blob with the provider's own key. - RotateSecret deleted and re-created; the re-create was dead, because the next read makes one anyway. Removed, and a second path to the same act is how two ends come to disagree. And one real fault: three places built a declaration, and the one behind `--json` predated credentials, so it silently produced a declaration missing them — a difference between what `plan` showed and what anything reading `--json` got. There is one path now.
152 lines
5.9 KiB
Go
152 lines
5.9 KiB
Go
package link
|
|
|
|
import (
|
|
"context"
|
|
"crypto/ed25519"
|
|
"crypto/rand"
|
|
"encoding/base64"
|
|
"errors"
|
|
"fmt"
|
|
|
|
"github.com/novox/mesh-control/internal/broker"
|
|
"github.com/novox/mesh-control/internal/identity"
|
|
"github.com/novox/mesh-control/internal/inventory"
|
|
)
|
|
|
|
// Enrolment is what actually happens when a node presents a token: the token is spent, the key is
|
|
// recorded, and the node gets its own queue.
|
|
//
|
|
// It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008)
|
|
// — it holds both grants and asks each for its part, which is what the process running them is
|
|
// for.
|
|
type Enrolment struct {
|
|
Inventory *inventory.Inventory
|
|
Identity *identity.Identity
|
|
Management *broker.Management
|
|
Broker broker.Broker
|
|
}
|
|
|
|
// Enrol spends the token and records what the node presented.
|
|
//
|
|
// Order matters and it is the order things become irreversible. The token is spent first, in a
|
|
// single statement that both finds and marks it, so two machines racing on one secret produce one
|
|
// winner. Only then is a key recorded — because recording a key for a node whose token turned out
|
|
// to be spent would leave the mesh believing a machine that never had the right to join.
|
|
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (EnrolReply, error) {
|
|
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
|
|
|
|
if len(public) != ed25519.PublicKeySize {
|
|
return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d",
|
|
len(public), ed25519.PublicKeySize)
|
|
}
|
|
|
|
node, err := e.Inventory.Redeem(ctx, secret)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
// From here the token is gone whatever happens next, so anything that fails leaves a node
|
|
// record with no live key — which is visible and fixable with a new token, where a spent
|
|
// token believed to be unspent is neither.
|
|
if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and the key could not be recorded, so %s has no identity and "+
|
|
"needs a new token: %w", node.Name, err)
|
|
}
|
|
|
|
key, err := e.Identity.Active(ctx)
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
|
|
reply := EnrolReply{
|
|
Accepted: true,
|
|
Node: node.Name,
|
|
Queue: QueueFor(node.Name),
|
|
Broker: e.Broker.Address,
|
|
Fingerprint: e.Broker.Fingerprint,
|
|
Signer: key.Public,
|
|
}
|
|
|
|
// The token's secret was the broker password up to this moment, which is what let this
|
|
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
|
|
// credential the node keeps for years is not the one that was pasted into a terminal.
|
|
if e.Management != nil {
|
|
password, err := freshPassword()
|
|
if err != nil {
|
|
return EnrolReply{}, err
|
|
}
|
|
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's broker password could not be replaced: %w",
|
|
node.Name, err)
|
|
}
|
|
reply.Password = password
|
|
}
|
|
|
|
// Recorded before the profile because the overlay is the first declaration this node will
|
|
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
|
|
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as
|
|
// briefly as possible.
|
|
// And the key its secrets are sealed to. Same reasoning as the overlay key below and one step
|
|
// stronger: without it the mesh cannot send this node a credential at all, and a node that
|
|
// enrolled without one will be refused a sealed file rather than quietly given none.
|
|
if request.SealingKey != "" {
|
|
if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's sealing key could not be recorded: %w",
|
|
node.Name, err)
|
|
}
|
|
}
|
|
if request.OverlayKey != "" {
|
|
if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil {
|
|
return EnrolReply{}, fmt.Errorf(
|
|
"the token was spent and %s's overlay key could not be recorded: %w", node.Name, err)
|
|
}
|
|
}
|
|
|
|
if profile != nil {
|
|
// Not fatal if it fails. The profile is what the control plane needs in order to decide
|
|
// what this machine should run, and it is reported again on every connection — so losing
|
|
// it here costs a decision that can be made later, not the enrolment.
|
|
_ = e.Inventory.RecordProfile(ctx, node.ID, profile)
|
|
}
|
|
return reply, nil
|
|
}
|
|
|
|
// freshPassword is the node's own broker credential from enrolment onward.
|
|
func freshPassword() (string, error) {
|
|
raw := make([]byte, 32)
|
|
if _, err := rand.Read(raw); err != nil {
|
|
return "", fmt.Errorf("cannot generate a broker password: %w", err)
|
|
}
|
|
return base64.RawURLEncoding.EncodeToString(raw), nil
|
|
}
|
|
|
|
var _ Enroller = Enrolment{}
|
|
|
|
// ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured.
|
|
var ErrNoBrokerManagement = errors.New("no broker management configured")
|
|
|
|
// Heard records what a node reported about itself.
|
|
//
|
|
// A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is
|
|
// its own account of its own machine, kept as a copy for recovery — so this writes it down and
|
|
// decides nothing from it.
|
|
func (e Enrolment) Heard(ctx context.Context, report Report) error {
|
|
if report.Node == "" {
|
|
return errors.New("a report named no node")
|
|
}
|
|
node, err := e.Inventory.NodeByName(ctx, report.Node)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
// A refusal, a failure, or a bare word that the node is there — none of them is an account of
|
|
// what the machine holds, so each moves last_seen and nothing else. Recording a partial list
|
|
// as though it were the whole would tell a rebuilding node to remove what it still has.
|
|
if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil {
|
|
return e.Inventory.Seen(ctx, node.ID)
|
|
}
|
|
return e.Inventory.RecordOwned(ctx, node.ID, report.Applied)
|
|
}
|