package link import ( "context" "crypto/ed25519" "crypto/rand" "crypto/sha256" "encoding/base64" "encoding/hex" "errors" "fmt" "log" "sort" "github.com/novox/mesh-controller/internal/broker" "github.com/novox/mesh-controller/internal/identity" "github.com/novox/mesh-controller/internal/inventory" ) // Enrolment is what actually happens when a node presents a token: the token is spent, the key is // recorded, and the node gets its own queue. // // It reaches across two contexts and reads neither one's store from the other (novox/hq ADR 0008) // — it holds both grants and asks each for its part, which is what the process running them is // for. type Enrolment struct { Inventory *inventory.Inventory Identity *identity.Identity Management *broker.Management Broker broker.Broker // OnNATS says the mesh's own traffic is on the bus being built, so a node's credential is // minted into the mesh's records and composed into the bus's user list rather than pushed // through a management call (novox/hq design 25 §4). // // **One bus, and a node gets a credential for exactly one** — refused at start if the // controller is told about both (broker.MustBeOneBus), because a node holding a credential for // each is one that could be half-moved, and nothing would say which half. OnNATS bool } // Enrol records what the node presented and spends the token. // // Order matters and it is the order things become irreversible (novox/hq issue 083). The token is // claimed first, in a single statement that both finds it and holds it for this presenter's key, so // two machines racing on one secret produce one holder. Then everything the node presented is // written — each write an overwrite, so an attempt interrupted by the store going away can be made // again by the same presenter. Then the token is spent. Last, the token's secret stops being the // node's broker password: done after the spend, because a password replaced by an attempt that // then failed would be one nobody holds, and the node could not even log in to ask again. func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (reply EnrolReply, err error) { secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile // A store that could not be asked right now, or a token another presenter holds for the // moment, is "not now": the node asks again with the same request (novox/hq issue 083). defer func() { if inventory.Unreachable(err) || errors.Is(err, inventory.ErrTokenInUse) || errors.Is(err, context.Canceled) { err = fmt.Errorf("%w: %w", ErrTryAgain, err) } }() if len(public) != ed25519.PublicKeySize { return EnrolReply{}, fmt.Errorf("a node presented a %d-byte key, and an identity is %d", len(public), ed25519.PublicKeySize) } // Claimed, not spent: the token is held for this presenter while the node is written, and // spent only as the last write. The node's identity lives in another database than the // token, so the two cannot be one transaction; a failure between them used to leave a spent // token and a node with no key, which the host — making new keys on every attempt — could not // recover from. Every write below overwrites, so an attempt made again is safe. // A proof that does not verify is refused outright: it was made with another key, or for // another request. One that verifies lets this presenter finish an enrolment whose token it // already spent — never a request the broker handed over a second time, which may already // have been answered. proven := false if len(request.Proof) > 0 { if !ed25519.Verify(public, EnrolProof(secret, public, request.OverlayKey, request.SealingKey, request.ServingKey), request.Proof) { return EnrolReply{}, errors.New("the enrolment's proof does not match the key it presents") } proven = true } by := claimant(public) node, err := e.Inventory.Claim(ctx, secret, by, proven && !request.Redelivered) if err != nil { return EnrolReply{}, err } if _, err := e.Identity.RecordNodeKey(ctx, node.ID, public); err != nil { return EnrolReply{}, fmt.Errorf("%s's key could not be recorded: %w", node.Name, err) } key, err := e.Identity.Active(ctx) if err != nil { return EnrolReply{}, err } reply = EnrolReply{ Accepted: true, Node: node.Name, Queue: QueueFor(node.Name), Broker: e.Broker.Address, Fingerprint: e.Broker.Fingerprint, Signer: key.Public, } // Recorded before the profile because the overlay is the first declaration this node will // receive, and without this key the mesh cannot compose one. A node enrolled with no overlay // key is a node the graph skips — an ordinary in-between state, and one worth leaving as // briefly as possible. // And the key its secrets are sealed to. Same reasoning as the overlay key below and one step // stronger: without it the mesh cannot send this node a credential at all, and a node that // enrolled without one will be refused a sealed file rather than quietly given none. // And the key it serves TLS with, so the mesh can certify its internal name. Public, so it is // recorded rather than sealed — the node keeps the half that matters. if request.ServingKey != "" { if err := e.Identity.RecordServingKey(ctx, node.ID, request.ServingKey); err != nil { return EnrolReply{}, fmt.Errorf( "%s's serving key could not be recorded: %w", node.Name, err) } } if request.SealingKey != "" { if err := e.Inventory.RecordSealingKey(ctx, node.ID, request.SealingKey); err != nil { return EnrolReply{}, fmt.Errorf( "%s's sealing key could not be recorded: %w", node.Name, err) } } if request.OverlayKey != "" { if err := e.Inventory.RecordOverlayKey(ctx, node.ID, request.OverlayKey); err != nil { return EnrolReply{}, fmt.Errorf( "%s's overlay key could not be recorded: %w", node.Name, err) } } // And the tunnel it found, whose key is the overlay key above (novox/hq ADR 0105). Recorded // before the token is spent for the same reason as the keys: the first declaration this node // receives is composed from it, and a hub enrolled without its tunnel would be placed at an // address of the mesh's choosing rather than the tunnel's. if request.Tunnel != nil { if request.Tunnel.PublicKey != request.OverlayKey { return EnrolReply{}, fmt.Errorf("%s presented a tunnel under key %s and an overlay key "+ "that is not it; a tunnel is taken over with its own key or not at all", node.Name, request.Tunnel.PublicKey) } peers := make([]inventory.TunnelPeer, 0, len(request.Tunnel.Peers)) for _, p := range request.Tunnel.Peers { peers = append(peers, inventory.TunnelPeer{PublicKey: p.PublicKey, Address: p.Address}) } if err := e.Inventory.RecordTunnel(ctx, node.ID, inventory.Tunnel{ Interface: request.Tunnel.Interface, Unit: request.Tunnel.Unit, Config: request.Tunnel.Config, Port: request.Tunnel.Port, Address: request.Tunnel.Address, Range: request.Tunnel.Range, PublicKey: request.Tunnel.PublicKey, Peers: peers, }); err != nil { return EnrolReply{}, fmt.Errorf("%s's found tunnel could not be recorded: %w", node.Name, err) } } // Spent once the node is complete in the store. if err := e.Inventory.Spend(ctx, secret, by); err != nil { return EnrolReply{}, fmt.Errorf("%s was written and its token could not be spent: %w", node.Name, err) } // The token's secret was the broker password up to this moment, which is what let this // connection exist at all. It is replaced now, so the one-time thing stays one-time and the // credential the node keeps for years is not the one that was pasted into a terminal. After // the spend and not before: a replaced password on an attempt that failed would be held by // nobody. If the broker will not take it now, the enrolment still stands — the node keeps // the token's secret as its password, which it is told, and which is said here. switch { case e.OnNATS: // **Minted into the mesh's records, not pushed to a server.** The bus's users are a file // the controller composes, so a credential becomes usable at the next composition rather // than at the moment it is made — and the plaintext is returned once, here, and then exists // only on the machine it was sealed to. // // The node reconnects as itself and may be refused until that composition reaches the // machine running the bus. That is what the host's reconnect backoff is for and it is // survivable by design (ADR 0004: disconnection is an ordinary situation); waiting for the // push here would hold an enrolment open for as long as a declaration takes to apply. password, err := e.Inventory.MintBusPassword(ctx, inventory.BusUser{ Username: broker.Principal{Kind: broker.KindNode, Node: node.Name}.Username(), Kind: inventory.BusNode, Node: node.Name, }) if err != nil { // Not fatal to the enrolment: the node is recorded and the token is spent, and a node // that keeps the token's secret is told so. Said loudly, because until this is minted // the machine has no credential of its own. log.Printf("%s is enrolled and the mesh could not mint its bus credential, so it keeps "+ "the token's secret as its password: %v", node.Name, err) } else { reply.Password = password } case e.Management != nil: password, err := freshPassword() if err != nil { return EnrolReply{}, err } if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil { log.Printf("%s is enrolled and its broker password could not be replaced, so it keeps "+ "the token's secret as its password: %v", node.Name, err) } else { reply.Password = password } } if profile != nil { // Not fatal if it fails. The profile is what the control plane needs in order to decide // what this machine should run, and it is reported again on every connection — so losing // it here costs a decision that can be made later, not the enrolment. _ = e.Inventory.RecordProfile(ctx, node.ID, profile) } return reply, nil } // rekey applies a verified rekey: the node's overlay key and tunnel are recorded as enrolment // would have recorded them, and a hub moves to the tunnel's address. func (e Enrolment) rekey(ctx context.Context, node inventory.Node, r Rekey) error { if r.Tunnel == nil || r.OverlayKey == "" { return fmt.Errorf("%s sent a rekey naming no tunnel or no key; refused", node.Name) } if e.Identity == nil { return fmt.Errorf("%s sent a rekey and this mesh has no identity store to verify it against", node.Name) } if err := e.Identity.VerifyNode(ctx, node.ID, RekeyProof(node.Name, r.Previous, r.OverlayKey, r.Tunnel), r.Proof); err != nil { return fmt.Errorf("%s's rekey is not signed by %s's identity key; refused: %w", node.Name, node.Name, err) } peers := make([]inventory.TunnelPeer, 0, len(r.Tunnel.Peers)) for _, p := range r.Tunnel.Peers { peers = append(peers, inventory.TunnelPeer{PublicKey: p.PublicKey, Address: p.Address}) } err := e.Inventory.Rekey(ctx, node.ID, r.Previous, r.OverlayKey, inventory.Tunnel{ Interface: r.Tunnel.Interface, Unit: r.Tunnel.Unit, Config: r.Tunnel.Config, Port: r.Tunnel.Port, Address: r.Tunnel.Address, Range: r.Tunnel.Range, PublicKey: r.Tunnel.PublicKey, Peers: peers, }) if err != nil { return fmt.Errorf("%s's rekey was not recorded: %w", node.Name, err) } log.Printf("%s took over the tunnel on %s: its overlay key is the tunnel's now", node.Name, r.Tunnel.Interface) return nil } // claimant names the key presenting a token, so a claim can be held for it alone. func claimant(public ed25519.PublicKey) string { sum := sha256.Sum256(public) return hex.EncodeToString(sum[:]) } // freshPassword is the node's own broker credential from enrolment onward. func freshPassword() (string, error) { raw := make([]byte, 32) if _, err := rand.Read(raw); err != nil { return "", fmt.Errorf("cannot generate a broker password: %w", err) } return base64.RawURLEncoding.EncodeToString(raw), nil } var _ Enroller = Enrolment{} // ErrNoBrokerManagement is returned when an account cannot be made because nothing was configured. var ErrNoBrokerManagement = errors.New("no broker management configured") // Heard records what a node reported about itself. // // A node states; the owning context writes (novox/hq ADR 0006). What a node says it applied is // its own account of its own machine, kept as a copy for recovery — so this writes it down and // decides nothing from it. // Outstanding is the declaration the mesh last sent a node, so a report about an older one is not // acted on (design 25 §3, window.go). // // **Here rather than on Listener.** A report is recorded by whatever keeps records, and a great // many things that record reports have no idea what was sent — every test in this package among // them. So the serving loop asks for this when the listener happens to be able to answer, and // where it cannot, a report has nothing to be stale against and is simply acted on. func (e Enrolment) Outstanding(ctx context.Context, node string) (string, error) { return e.Inventory.Outstanding(ctx, node) } func (e Enrolment) Heard(ctx context.Context, report Report) (err error) { // A store that could not be asked right now is said as such, so the report is kept for // another attempt rather than acknowledged and lost (novox/hq issue 082). defer func() { if inventory.Unreachable(err) { err = fmt.Errorf("%w: %w", ErrTryAgain, err) } }() if report.Node == "" { return errors.New("a report named no node") } node, err := e.Inventory.NodeByName(ctx, report.Node) if err != nil { return err } // What an adopted node holds, which firewall it found, and what is reachable on it (novox/hq // ADR 0100). Recorded whenever a report carries any of it — a node reports these on its own // schedule, when what it holds changes, not only after an apply — and never cleared by a // report that carries none, which is every bare word that the node is there. An adopted node // always names its firewall, so a report from one replaces all three, emptied held included. if len(report.Held) > 0 || report.Firewall != "" || len(report.Reachable) > 0 { held := make([]inventory.Held, 0, len(report.Held)) for _, h := range report.Held { held = append(held, inventory.Held{ID: h.ID, Module: h.Module, Kind: h.Kind, Target: h.Target, Since: h.Since, Changed: h.Changed, Kept: h.Kept}) } reachable := make([]inventory.Reach, 0, len(report.Reachable)) for _, r := range report.Reachable { reachable = append(reachable, inventory.Reach{Protocol: r.Protocol, Address: r.Address, Port: r.Port, By: r.By, Published: r.Published, ContainerPort: r.ContainerPort}) } if err := e.Inventory.RecordAdoption(ctx, node.ID, held, report.Firewall, reachable); err != nil { return err } } // What it says about the tunnel it carried (novox/hq ADR 0105), whenever it says it. if report.Tunnel != nil { if err := e.Inventory.RecordCarriedTunnel(ctx, node.ID, inventory.Carried{ Interface: report.Tunnel.Interface, Port: report.Tunnel.Port, Range: report.Tunnel.Range, Peers: report.Tunnel.Peers, State: report.Tunnel.State, Note: report.Tunnel.Note, Kept: report.Tunnel.Kept, }); err != nil { return err } } // A node taking a found tunnel's key after enrolment (novox/hq ADR 0105). Verified against the // node's live identity key before anything is written: the broker account authenticates the // connection, the signature proves the node itself said it. Refused outright when the proof // does not verify or is stale — a refusal, not "not now", so the node hears why. if report.Rekey != nil { if err := e.rekey(ctx, node, *report.Rekey); err != nil { return err } return e.Inventory.Seen(ctx, node.ID) } // A bare word that a node is there is not an account of what the machine did or holds: it // moves last_seen and touches nothing else. This arrives every minute (link.AliveEvery), // while a real report is rare, so recording it as one would overwrite the node's last real // apply with an empty one — wiping the declaration digest that decides whether the node is // current, the carried ports a push assigns around, and the clean-or-failed outcome — and a // node that had just caught up would read as behind within the minute. The alive path calls // this with only a node name; a real report always carries an account (something applied, or // a refusal, or a failure), so those are the reports that get written down. if report.Applied == nil && report.Refused == "" && len(report.Failed) == 0 { if report.Superseded != "" { log.Printf("%s set aside declaration %s for the newer %s", report.Node, report.Declared, report.Superseded) } return e.Inventory.Seen(ctx, node.ID) } // What it did is kept whichever way it went. Until this, a refusal or a failure moved // last_seen and the reason went to a log line, so "which machine is not doing what it was // told" had no answer the next morning — which is the question a mesh exists to answer. doing := inventory.Doing{ Outcome: inventory.OutcomeApplied, Refused: report.Refused, Applied: len(report.Applied), Declared: report.Declared, } switch { case report.Refused != "": doing.Outcome = inventory.OutcomeRefused case len(report.Failed) > 0: doing.Outcome = inventory.OutcomeFailed } for id, why := range report.Failed { doing.Failed = append(doing.Failed, inventory.FailedResource{ID: id, Error: why}) } // Ordered, so two readings of one failure are the same reading. sort.Slice(doing.Failed, func(i, j int) bool { return doing.Failed[i].ID < doing.Failed[j].ID }) // And what that machine says it already holds, so a port is assigned around it rather than // on top of it (novox/hq ADR 0038). Kept even when the declaration was refused: what the // machine carries is true regardless of what it thought of the last thing it was sent. if err := e.Inventory.RecordCarried(ctx, report.Node, report.Carried); err != nil { return err } if err := e.Inventory.RecordDoing(ctx, node.ID, doing); err != nil { return err } // A refusal, a failure, or a bare word that the node is there — none of them is an account of // what the machine holds, so each moves last_seen and nothing else. Recording a partial list // as though it were the whole would tell a rebuilding node to remove what it still has. if report.Refused != "" || len(report.Failed) > 0 || report.Applied == nil { return e.Inventory.Seen(ctx, node.ID) } return e.Inventory.RecordOwned(ctx, node.ID, report.Applied) }