The host's inbound behind a seam, with both transports
The outbound half went behind `Bus` and a node's two statements stopped naming a transport. This is the other half, and where the transport reached furthest: the run loop selected on a channel of the client library's own delivery type, so every part of holding a node in its mesh knew which bus it was on. `Link` is dialling, hearing and saying in one interface, because dialling is where the transport is chosen and choosing it twice is how one half of a node ends up on a different bus from the other. `Declaration` has one way of being done rather than two: a declaration set aside for a newer one is settled exactly as an applied one is, on both buses, and the difference is a fact the report carries. Four things this settled. **The host declares nothing on the new bus.** On the bus the mesh has it declares its own queue, because a queue that is not there means a node that hears nothing. Here it binds to a consumer the mesh made when the node enrolled, and a missing one is said as the mesh's to answer rather than quietly created with whatever this client happens to default to. **The pin is easier here than in the tool runtime, not harder.** The Go client takes a *tls.Config, so the same PinnedConfig with the same VerifyPeerCertificate does the work — the subject-alternative-name constraint recorded against the runtime's client is that client's, because it takes PEM strings with no verify hook. A host checks the fingerprint and nothing else. **Binding needs the subject as well as the consumer.** An empty subject is refused rather than taken to mean "whatever that consumer delivers", which the server said plainly and only when asked. **Reconnection stays the caller's.** Hold already decides when to try again and how long to wait; a client reconnecting underneath it would make that reasoning a duplicate of the library's. The drain keeps its live half and loses its catch-up half, as it said it would: verified that three declarations pushed to an absent node leave one on the stream, and it is the newest. One test-harness lesson worth the comment it got: delete-then-add is not a reset. A test that did that inherited the previous test's messages, and the symptom was a declaration counted as delivered twice — which reads as a redelivery bug in the code under test rather than as a dirty stream.
This commit is contained in:
+42
-108
@@ -6,10 +6,7 @@ import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"time"
|
||||
|
||||
amqp "github.com/rabbitmq/amqp091-go"
|
||||
)
|
||||
|
||||
// ErrForged is what a node returns for a declaration whose signature is not the mesh's.
|
||||
@@ -33,6 +30,10 @@ type Membership struct {
|
||||
Fingerprint string
|
||||
Password string
|
||||
Signer ed25519.PublicKey
|
||||
// Transport is which bus this node speaks (hearing.go). Empty is the one the mesh runs on
|
||||
// today, which is every node until the rollout — so a membership recorded before any of this
|
||||
// existed reads as correct rather than as unset.
|
||||
Transport string
|
||||
}
|
||||
|
||||
// Applier is what the host does with a declaration that has been proved to come from the mesh.
|
||||
@@ -190,108 +191,62 @@ func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout
|
||||
if say == nil {
|
||||
say = func(string) {}
|
||||
}
|
||||
config, err := PinnedConfig(m.Fingerprint)
|
||||
link, err := Open(ctx, m, timeout)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer link.Close()
|
||||
|
||||
dsn := fmt.Sprintf("amqps://%s:%s@%s/",
|
||||
url.QueryEscape(m.Node), url.QueryEscape(m.Password), m.Broker)
|
||||
conn, err := amqp.DialConfig(dsn, amqp.Config{
|
||||
TLSClientConfig: config,
|
||||
Dial: amqp.DefaultDial(timeout),
|
||||
// Kept short so a node that has silently lost its route notices, rather than holding a
|
||||
// connection the broker forgot about and believing it is still in the mesh.
|
||||
Heartbeat: 10 * time.Second,
|
||||
})
|
||||
if err != nil {
|
||||
if errors.Is(err, ErrWrongCertificate) {
|
||||
return err
|
||||
}
|
||||
return fmt.Errorf("cannot reach the broker at %s: %w", m.Broker, err)
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
channel, err := conn.Channel()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer channel.Close()
|
||||
|
||||
queue := QueueFor(m.Node)
|
||||
if _, err := channel.QueueDeclare(queue, true, false, false, false, nil); err != nil {
|
||||
return fmt.Errorf("cannot declare this node's queue %s: %w", queue, err)
|
||||
}
|
||||
|
||||
// Applying is one at a time — two at once would race on the same filesystem — but SEEING is
|
||||
// not: with a prefetch of one the host could never know that a newer declaration was already
|
||||
// waiting, and so applied every one of a backlog in turn, at the better part of a minute each,
|
||||
// becoming things nobody wanted any more (novox/hq issue 031). A window of unacknowledged
|
||||
// deliveries lets it drain to the newest; each declaration still survives a restart on the
|
||||
// broker until it is acknowledged, which happens only after it is applied or set aside.
|
||||
if err := channel.Qos(drainDepth, 0, false); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
deliveries, err := channel.ConsumeWithContext(ctx, queue, "", false, false, false, false, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// Said, because it is the event anybody watching actually wants. Without it a node logs
|
||||
// every failure and nothing on success, so a log full of "trying again" and then silence
|
||||
// reads as still broken when it means the opposite.
|
||||
say("in the mesh, consuming " + queue)
|
||||
// Said, because it is the event anybody watching actually wants. Without it a node logs every
|
||||
// failure and nothing on success, so a log full of "trying again" and then silence reads as
|
||||
// still broken when it means the opposite.
|
||||
say("in the mesh, hearing what this node should be")
|
||||
|
||||
// A word every so often, so the mesh can tell a node that is quiet from one that is gone.
|
||||
// Cheap on purpose: it carries a name and nothing else, because anything more would be a
|
||||
// report, and reports are rare where this is constant.
|
||||
beat := time.NewTicker(AliveEvery)
|
||||
defer beat.Stop()
|
||||
publishAlive(ctx, OverCurrent{Channel: channel}, m, say, timeout)
|
||||
publishAlive(ctx, link, m, say, timeout)
|
||||
|
||||
closed := conn.NotifyClose(make(chan *amqp.Error, 1))
|
||||
|
||||
// Published mandatory, so the broker hands back anything it cannot route rather than
|
||||
// dropping it. Without this a report goes to an exchange with no matching binding, the
|
||||
// publisher is told nothing, and the mesh believes this node never answered while the node
|
||||
// believes it did — which is what happened when `report` was left unbound on the other side.
|
||||
returned := channel.NotifyReturn(make(chan amqp.Return, 4))
|
||||
go func() {
|
||||
for r := range returned {
|
||||
say(fmt.Sprintf("the broker could not route this node's %s: %s (%d %s)",
|
||||
r.RoutingKey, r.Exchange, r.ReplyCode, r.ReplyText))
|
||||
}
|
||||
}()
|
||||
declarations := link.Declarations()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil
|
||||
case <-beat.C:
|
||||
publishAlive(ctx, OverCurrent{Channel: channel}, m, say, timeout)
|
||||
publishAlive(ctx, link, m, say, timeout)
|
||||
case unasked := <-outbox:
|
||||
// Said without having been asked: a reconcile found what an adopted node holds, or
|
||||
// its firewall, changed since it last said.
|
||||
published := publishReport(ctx, OverCurrent{Channel: channel}, m, unasked.Report, say, timeout)
|
||||
published := publishReport(ctx, link, m, unasked.Report, say, timeout)
|
||||
if unasked.Done != nil {
|
||||
unasked.Done(published)
|
||||
}
|
||||
case reason := <-closed:
|
||||
return fmt.Errorf("the link closed: %v", reason)
|
||||
case delivery, ok := <-deliveries:
|
||||
case reason := <-link.Lost():
|
||||
return reason
|
||||
case declaration, ok := <-declarations:
|
||||
if !ok {
|
||||
return errors.New("the broker stopped delivering")
|
||||
// The link's own reason, when it has managed to say one: "stopped delivering" on
|
||||
// its own says nothing about why, and why is the whole of what an operator wants.
|
||||
select {
|
||||
case reason := <-link.Lost():
|
||||
return reason
|
||||
default:
|
||||
return errors.New("the mesh stopped sending this node declarations")
|
||||
}
|
||||
}
|
||||
// Whatever else is already waiting supersedes this one. Each set-aside declaration
|
||||
// is reported as such, then acknowledged unapplied.
|
||||
delivery, superseded := newest(deliveries, delivery, drainWindow)
|
||||
// Whatever else is already waiting supersedes this one. Each set-aside declaration is
|
||||
// reported as such, then settled unapplied.
|
||||
declaration, superseded := newest(declarations, declaration, drainWindow)
|
||||
for _, old := range superseded {
|
||||
say("set aside a declaration: a newer one arrived with it")
|
||||
publishReport(ctx, OverCurrent{Channel: channel}, m, Report{Node: m.Node, Declared: declaredIn(old.Body),
|
||||
Superseded: declaredIn(delivery.Body)}, say, timeout)
|
||||
_ = old.Ack(false)
|
||||
publishReport(ctx, link, m, Report{Node: m.Node, Declared: declaredIn(old.Body()),
|
||||
Superseded: declaredIn(declaration.Body())}, say, timeout)
|
||||
_ = old.Handled()
|
||||
}
|
||||
report := handle(ctx, m, apply, delivery)
|
||||
report := handleBody(ctx, m, declaration.Body(), apply)
|
||||
switch {
|
||||
case report.Refused != "":
|
||||
say("refused a declaration: " + report.Refused)
|
||||
@@ -300,12 +255,12 @@ func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout
|
||||
default:
|
||||
say(fmt.Sprintf("applied %d resource(s)", len(report.Applied)))
|
||||
}
|
||||
publishReport(ctx, OverCurrent{Channel: channel}, m, report, say, timeout)
|
||||
// Acknowledged after the report is published. A node that dies between applying and
|
||||
// reporting leaves the declaration on the broker and applies it again on return,
|
||||
publishReport(ctx, link, m, report, say, timeout)
|
||||
// Settled after the report is published. A node that dies between applying and
|
||||
// reporting leaves the declaration with the mesh and applies it again on return,
|
||||
// which is safe because applying is reconciliation — it converges rather than
|
||||
// repeating.
|
||||
_ = delivery.Ack(false)
|
||||
_ = declaration.Handled()
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -333,12 +288,12 @@ const (
|
||||
// three deliveries, whatever the stream later retains. So this is narrowed at the rollout, not
|
||||
// deleted — and saying which half goes is worth more than a note that it "can probably be
|
||||
// removed", which is how a load-bearing window gets deleted by somebody in a hurry.
|
||||
func newest(deliveries <-chan amqp.Delivery, first amqp.Delivery, window time.Duration) (amqp.Delivery, []amqp.Delivery) {
|
||||
func newest(arriving <-chan Declaration, first Declaration, window time.Duration) (Declaration, []Declaration) {
|
||||
latest := first
|
||||
var superseded []amqp.Delivery
|
||||
var superseded []Declaration
|
||||
for {
|
||||
select {
|
||||
case next, ok := <-deliveries:
|
||||
case next, ok := <-arriving:
|
||||
if !ok {
|
||||
return latest, superseded
|
||||
}
|
||||
@@ -367,10 +322,6 @@ func declaredIn(body []byte) string {
|
||||
return d.Declared
|
||||
}
|
||||
|
||||
func handle(ctx context.Context, m Membership, apply Applier, delivery amqp.Delivery) Report {
|
||||
return handleBody(ctx, m, delivery.Body, apply)
|
||||
}
|
||||
|
||||
// handleBody is the whole of deciding whether to trust a message, separated from the broker so it
|
||||
// can be tested as the security check it is rather than as message plumbing.
|
||||
func handleBody(ctx context.Context, m Membership, body []byte, apply Applier) Report {
|
||||
@@ -393,30 +344,13 @@ func handleBody(ctx context.Context, m Membership, body []byte, apply Applier) R
|
||||
// report a command makes rather than the running host — a rekey (novox/hq ADR 0105). The same
|
||||
// account, the same pinned certificate and the same exchange as the running host's reports.
|
||||
func Publish(ctx context.Context, m Membership, report Report, timeout time.Duration) error {
|
||||
config, err := PinnedConfig(m.Fingerprint)
|
||||
link, err := Open(ctx, m, timeout)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dsn := fmt.Sprintf("amqps://%s:%s@%s/",
|
||||
url.QueryEscape(m.Node), url.QueryEscape(m.Password), m.Broker)
|
||||
conn, err := amqp.DialConfig(dsn, amqp.Config{
|
||||
TLSClientConfig: config,
|
||||
Dial: amqp.DefaultDial(timeout),
|
||||
})
|
||||
if err != nil {
|
||||
if errors.Is(err, ErrWrongCertificate) {
|
||||
return err
|
||||
}
|
||||
return fmt.Errorf("cannot reach the broker at %s: %w", m.Broker, err)
|
||||
}
|
||||
defer conn.Close()
|
||||
channel, err := conn.Channel()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer channel.Close()
|
||||
defer link.Close()
|
||||
var said string
|
||||
if !publishReport(ctx, OverCurrent{Channel: channel}, m, report, func(s string) { said = s }, timeout) {
|
||||
if !publishReport(ctx, link, m, report, func(s string) { said = s }, timeout) {
|
||||
return errors.New(said)
|
||||
}
|
||||
return nil
|
||||
|
||||
Reference in New Issue
Block a user