The host's outbound behind a seam, with both transports
Step 3.5's first half, mirroring the controller's. A host says exactly two things unprompted, and the difference between them is the whole interface: a report must arrive, and a heartbeat must not be insisted on. So a report goes through JetStream — it is the message the store-window guarantee is about — and a heartbeat stays on core, because a heartbeat in a stream is the mesh's least valuable message competing for retention with its most valuable. The host still imports nothing of the mesh's own (ADR 0005): this is its own interface over its own libraries. It agrees with the controller because a fixture holds both to one envelope, which is the only agreement that survives two repositories. Also recorded, where the next person reads it rather than in a plan: the "newest wins" window narrows at the rollout and does not disappear. Last- per-subject makes the catch-up half the stream's, and sequence orders them definitively — but three pushes to a connected node are still three deliveries. Saying which half goes is worth more than "can probably be removed", which is how a load-bearing window gets deleted in a hurry.
This commit is contained in:
+22
-12
@@ -247,7 +247,7 @@ func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout
|
||||
// report, and reports are rare where this is constant.
|
||||
beat := time.NewTicker(AliveEvery)
|
||||
defer beat.Stop()
|
||||
publishAlive(ctx, channel, m, say, timeout)
|
||||
publishAlive(ctx, OverCurrent{Channel: channel}, m, say, timeout)
|
||||
|
||||
closed := conn.NotifyClose(make(chan *amqp.Error, 1))
|
||||
|
||||
@@ -268,11 +268,11 @@ func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout
|
||||
case <-ctx.Done():
|
||||
return nil
|
||||
case <-beat.C:
|
||||
publishAlive(ctx, channel, m, say, timeout)
|
||||
publishAlive(ctx, OverCurrent{Channel: channel}, m, say, timeout)
|
||||
case unasked := <-outbox:
|
||||
// Said without having been asked: a reconcile found what an adopted node holds, or
|
||||
// its firewall, changed since it last said.
|
||||
published := publishReport(ctx, channel, m, unasked.Report, say, timeout)
|
||||
published := publishReport(ctx, OverCurrent{Channel: channel}, m, unasked.Report, say, timeout)
|
||||
if unasked.Done != nil {
|
||||
unasked.Done(published)
|
||||
}
|
||||
@@ -287,7 +287,7 @@ func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout
|
||||
delivery, superseded := newest(deliveries, delivery, drainWindow)
|
||||
for _, old := range superseded {
|
||||
say("set aside a declaration: a newer one arrived with it")
|
||||
publishReport(ctx, channel, m, Report{Node: m.Node, Declared: declaredIn(old.Body),
|
||||
publishReport(ctx, OverCurrent{Channel: channel}, m, Report{Node: m.Node, Declared: declaredIn(old.Body),
|
||||
Superseded: declaredIn(delivery.Body)}, say, timeout)
|
||||
_ = old.Ack(false)
|
||||
}
|
||||
@@ -300,7 +300,7 @@ func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout
|
||||
default:
|
||||
say(fmt.Sprintf("applied %d resource(s)", len(report.Applied)))
|
||||
}
|
||||
publishReport(ctx, channel, m, report, say, timeout)
|
||||
publishReport(ctx, OverCurrent{Channel: channel}, m, report, say, timeout)
|
||||
// Acknowledged after the report is published. A node that dies between applying and
|
||||
// reporting leaves the declaration on the broker and applies it again on return,
|
||||
// which is safe because applying is reconciliation — it converges rather than
|
||||
@@ -321,6 +321,18 @@ const (
|
||||
// newest takes what is already waiting behind `first` and returns the last of them to apply, and
|
||||
// the rest to set aside. It waits `window` for a straggler after each arrival and no longer: a
|
||||
// declaration in flight from the mesh arrives within that; one that does not is the next push.
|
||||
//
|
||||
// **Its job narrows once declarations are state rather than messages, and does not disappear.**
|
||||
// On the bus being built, a declaration is last-per-subject (novox/hq design 29 §4), so a node
|
||||
// that was away receives exactly the current one instead of a queue of superseded ones — the
|
||||
// catch-up half of what this does is then the stream's. And a stream sequence orders them
|
||||
// definitively, where this window only infers order from arrival time, which is the wire-level
|
||||
// answer to novox/hq issue 107.
|
||||
//
|
||||
// What remains is the live case: three pushes in quick succession to a *connected* node are
|
||||
// three deliveries, whatever the stream later retains. So this is narrowed at the rollout, not
|
||||
// deleted — and saying which half goes is worth more than a note that it "can probably be
|
||||
// removed", which is how a load-bearing window gets deleted by somebody in a hurry.
|
||||
func newest(deliveries <-chan amqp.Delivery, first amqp.Delivery, window time.Duration) (amqp.Delivery, []amqp.Delivery) {
|
||||
latest := first
|
||||
var superseded []amqp.Delivery
|
||||
@@ -404,13 +416,13 @@ func Publish(ctx context.Context, m Membership, report Report, timeout time.Dura
|
||||
}
|
||||
defer channel.Close()
|
||||
var said string
|
||||
if !publishReport(ctx, channel, m, report, func(s string) { said = s }, timeout) {
|
||||
if !publishReport(ctx, OverCurrent{Channel: channel}, m, report, func(s string) { said = s }, timeout) {
|
||||
return errors.New(said)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func publishReport(ctx context.Context, channel *amqp.Channel, m Membership, report Report,
|
||||
func publishReport(ctx context.Context, bus Bus, m Membership, report Report,
|
||||
say Announce, timeout time.Duration) bool {
|
||||
report.Node = m.Node
|
||||
body, err := json.Marshal(report)
|
||||
@@ -424,8 +436,7 @@ func publishReport(ctx context.Context, channel *amqp.Channel, m Membership, rep
|
||||
// Said rather than swallowed. A report that fails to publish leaves the mesh believing this
|
||||
// node never answered, while the node believes it did — and the two would go on disagreeing
|
||||
// with nothing anywhere saying so. That shape of fault is the one this project keeps finding.
|
||||
if err := channel.PublishWithContext(publish, Exchange, KeyReport, true, false,
|
||||
amqp.Publishing{ContentType: "application/json", Body: body}); err != nil {
|
||||
if err := bus.Report(publish, m.Node, body); err != nil {
|
||||
say(fmt.Sprintf("applied, and could not tell the mesh: %v", err))
|
||||
return false
|
||||
}
|
||||
@@ -433,7 +444,7 @@ func publishReport(ctx context.Context, channel *amqp.Channel, m Membership, rep
|
||||
}
|
||||
|
||||
// publishAlive says this node is here, and nothing else.
|
||||
func publishAlive(ctx context.Context, channel *amqp.Channel, m Membership, say Announce,
|
||||
func publishAlive(ctx context.Context, bus Bus, m Membership, say Announce,
|
||||
timeout time.Duration) {
|
||||
body, err := json.Marshal(Alive{Node: m.Node})
|
||||
if err != nil {
|
||||
@@ -444,8 +455,7 @@ func publishAlive(ctx context.Context, channel *amqp.Channel, m Membership, say
|
||||
// Not mandatory, unlike a report. Losing one is nothing: the next is a minute away, and the
|
||||
// mesh is reading a gap rather than counting arrivals. Insisting on delivery would turn a
|
||||
// harmless miss into a logged failure every minute.
|
||||
if err := channel.PublishWithContext(publish, Exchange, KeyAlive, false, false,
|
||||
amqp.Publishing{ContentType: "application/json", Body: body}); err != nil {
|
||||
if err := bus.Alive(publish, m.Node, body); err != nil {
|
||||
say("could not tell the mesh this node is here: " + err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user