Files
mesh-host/internal/link/hearing_nats_test.go
T
jschoubben 6e208f7b3e The host's inbound behind a seam, with both transports
The outbound half went behind `Bus` and a node's two statements stopped naming a
transport. This is the other half, and where the transport reached furthest: the
run loop selected on a channel of the client library's own delivery type, so
every part of holding a node in its mesh knew which bus it was on.

`Link` is dialling, hearing and saying in one interface, because dialling is
where the transport is chosen and choosing it twice is how one half of a node
ends up on a different bus from the other. `Declaration` has one way of being
done rather than two: a declaration set aside for a newer one is settled exactly
as an applied one is, on both buses, and the difference is a fact the report
carries.

Four things this settled.

**The host declares nothing on the new bus.** On the bus the mesh has it declares
its own queue, because a queue that is not there means a node that hears
nothing. Here it binds to a consumer the mesh made when the node enrolled, and a
missing one is said as the mesh's to answer rather than quietly created with
whatever this client happens to default to.

**The pin is easier here than in the tool runtime, not harder.** The Go client
takes a *tls.Config, so the same PinnedConfig with the same VerifyPeerCertificate
does the work — the subject-alternative-name constraint recorded against the
runtime's client is that client's, because it takes PEM strings with no verify
hook. A host checks the fingerprint and nothing else.

**Binding needs the subject as well as the consumer.** An empty subject is
refused rather than taken to mean "whatever that consumer delivers", which the
server said plainly and only when asked.

**Reconnection stays the caller's.** Hold already decides when to try again and
how long to wait; a client reconnecting underneath it would make that reasoning
a duplicate of the library's.

The drain keeps its live half and loses its catch-up half, as it said it would:
verified that three declarations pushed to an absent node leave one on the
stream, and it is the newest.

One test-harness lesson worth the comment it got: delete-then-add is not a reset.
A test that did that inherited the previous test's messages, and the symptom was
a declaration counted as delivered twice — which reads as a redelivery bug in the
code under test rather than as a dirty stream.
2026-09-27 01:25:01 +02:00

229 lines
7.7 KiB
Go

package link
import (
"context"
"crypto/ed25519"
"encoding/json"
"fmt"
"os"
"testing"
"time"
"github.com/nats-io/nats.go"
)
// The host's link against a real server, because every claim here is about one.
//
// Whether binding to a consumer the host did not create works, whether a declaration on the node's
// own subject arrives, whether acknowledging it removes it from the consumer's pending — none of
// that can be reasoned out, and the first two are the ones that would leave a node silently hearing
// nothing:
//
// docker run -d --rm --name t -p 14223:4222 nats:2.10-alpine -js
// MESH_TEST_NATS=nats://127.0.0.1:14223 go test ./internal/link/ -run TestNats
func aBus(t *testing.T) (*nats.Conn, nats.JetStreamContext) {
t.Helper()
url := os.Getenv("MESH_TEST_NATS")
if url == "" {
t.Skip("MESH_TEST_NATS unset")
}
conn, err := nats.Connect(url)
if err != nil {
t.Fatal(err)
}
t.Cleanup(conn.Close)
js, err := conn.JetStream()
if err != nil {
t.Fatal(err)
}
// **Ensured and purged, not deleted and recreated.** Delete-then-add looked like a reset and is
// not one: a test that did that inherited the previous test's messages, and the symptom was a
// declaration counted as delivered twice — which reads as a redelivery bug in the code under
// test rather than as a dirty stream. Purge is defined to empty a stream; recreating one is a
// race with the server's own teardown.
for _, want := range []*nats.StreamConfig{
{Name: "NODES", Subjects: []string{"mesh.node.*.declare"}, MaxMsgsPerSubject: 1},
{Name: "CONTROL", Subjects: []string{"mesh.control.*.report", "mesh.control.enrol"},
Retention: nats.WorkQueuePolicy},
} {
if _, err := js.StreamInfo(want.Name); err != nil {
if _, err := js.AddStream(want); err != nil {
t.Fatal(err)
}
}
if err := js.PurgeStream(want.Name); err != nil {
t.Fatal(err)
}
}
return conn, js
}
// theMeshMakes is the consumer the controller creates when a node enrols. Made here by the test
// because the host may not: its account reaches no part of the JetStream API, which is the whole
// reason this binds rather than subscribes.
//
// Removed afterwards, and each test names its own node: two tests sharing a consumer name share its
// delivery count and its pending list, and the first thing that goes wrong reads as a fault in the
// host rather than in the test beside it.
func theMeshMakes(t *testing.T, js nats.JetStreamContext, node string) {
t.Helper()
t.Cleanup(func() { _ = js.DeleteConsumer("NODES", node) })
if _, err := js.AddConsumer("NODES", &nats.ConsumerConfig{
Durable: node,
FilterSubject: DeclareSubject(node),
AckPolicy: nats.AckExplicitPolicy,
AckWait: 300 * time.Second,
DeliverSubject: "_DELIVER." + node,
}); err != nil {
t.Fatal(err)
}
}
func signedBy(t *testing.T, key ed25519.PrivateKey, declaration []byte) []byte {
t.Helper()
body, err := json.Marshal(Signed{
Declaration: declaration, Signature: ed25519.Sign(key, declaration),
})
if err != nil {
t.Fatal(err)
}
return body
}
// A declaration on this node's own subject reaches the host, is applied, and acknowledging it
// empties the consumer — which is what tells the mesh the node has it.
func TestNatsADeclarationReachesTheHostAndIsSettled(t *testing.T) {
conn, js := aBus(t)
const node = "settling"
theMeshMakes(t, js, node)
public, private, _ := ed25519.GenerateKey(nil)
m := Membership{Node: node, Signer: public}
// Dialled directly rather than through Open: the test server has no TLS, and what is being
// checked is the subscription and the settling, not the pin — which PinnedConfig owns and its
// own tests cover.
l := &natsLink{conn: conn, js: js, node: node,
arrived: make(chan Declaration, drainDepth), lost: make(chan error, 1)}
feed := make(chan *nats.Msg, drainDepth)
sub, err := js.ChanSubscribe(DeclareSubject(node), feed, nats.Bind("NODES", node))
if err != nil {
t.Fatalf("the host could not bind to the consumer the mesh made for it: %v", err)
}
defer func() { _ = sub.Unsubscribe() }()
go func() {
for msg := range feed {
l.arrived <- natsDeclaration{msg}
}
}()
if _, err := js.Publish(DeclareSubject(node),
signedBy(t, private, []byte(`{"declared":"d1"}`))); err != nil {
t.Fatal(err)
}
select {
case d := <-l.Declarations():
report := handleBody(context.Background(), m, d.Body(),
func(context.Context, []byte, []byte) Report {
return Report{Applied: []string{"store"}}
})
if report.Refused != "" {
t.Fatalf("a declaration the mesh signed was refused: %s", report.Refused)
}
if err := d.Handled(); err != nil {
t.Fatalf("the node could not acknowledge its own declaration: %v", err)
}
case <-time.After(8 * time.Second):
t.Fatal("no declaration reached the host")
}
// **Nothing pending is the property**; a delivery count is not. Delivery is at-least-once by
// design, so pinning "delivered exactly once" would be asserting something the mesh does not
// rely on. What matters is that the acknowledgement landed, so the mesh can tell the node has
// it — and that no redelivery was needed to get there, which is what would say the node was
// too slow to answer for its own ack wait.
deadline := time.Now().Add(5 * time.Second)
var last string
for time.Now().Before(deadline) {
info, err := js.ConsumerInfo("NODES", node)
switch {
case err != nil:
last = err.Error()
case info.NumAckPending == 0 && info.NumRedelivered == 0:
return
default:
last = fmt.Sprintf("pending %d, redelivered %d", info.NumAckPending, info.NumRedelivered)
}
time.Sleep(20 * time.Millisecond)
}
t.Fatalf("the declaration was not settled, so the mesh cannot tell the node has it: %s", last)
}
// **A node that was away gets exactly the current declaration and nothing older.** Three pushed
// while nothing is listening leave one on the stream, and it is the newest — the wire-level answer
// to novox/hq issue 107, and the half of the drain that stops being the host's problem.
func TestNatsANodeThatWasAwayGetsOnlyTheNewest(t *testing.T) {
_, js := aBus(t)
const node = "returning"
_, private, _ := ed25519.GenerateKey(nil)
for _, id := range []string{"d1", "d2", "d3"} {
if _, err := js.Publish(DeclareSubject(node),
signedBy(t, private, []byte(`{"declared":"`+id+`"}`))); err != nil {
t.Fatal(err)
}
}
info, err := js.StreamInfo("NODES")
if err != nil {
t.Fatal(err)
}
if info.State.Msgs != 1 {
t.Fatalf("%d declarations survived for one node; a node that was away would apply a backlog "+
"of things nobody wants any more", info.State.Msgs)
}
theMeshMakes(t, js, node)
feed := make(chan *nats.Msg, drainDepth)
sub, err := js.ChanSubscribe(DeclareSubject(node), feed, nats.Bind("NODES", node))
if err != nil {
t.Fatal(err)
}
defer func() { _ = sub.Unsubscribe() }()
select {
case msg := <-feed:
if declaredIn(msg.Data) != "d3" {
t.Fatalf("the node was given %q rather than the newest", declaredIn(msg.Data))
}
case <-time.After(8 * time.Second):
t.Fatal("the node that was away was given nothing")
}
}
// A report goes through the stream and a heartbeat does not: the one that must survive the
// controller's store restarting is kept, and the one that must not is not.
func TestNatsAReportIsKeptAndAHeartbeatIsNot(t *testing.T) {
conn, js := aBus(t)
bus := OverNATS{Conn: conn, JS: js}
ctx := context.Background()
body, _ := json.Marshal(Report{Node: "anchor", Declared: "d1"})
if err := bus.Report(ctx, "anchor", body); err != nil {
t.Fatal(err)
}
beat, _ := json.Marshal(Alive{Node: "anchor"})
if err := bus.Alive(ctx, "anchor", beat); err != nil {
t.Fatal(err)
}
info, err := js.StreamInfo("CONTROL")
if err != nil {
t.Fatal(err)
}
if info.State.Msgs != 1 {
t.Fatalf("%d messages were kept; a report must be and a heartbeat must not", info.State.Msgs)
}
}