Derive a seat's stream and a module's consumer, and wire JetStream

Task 3.9's other half and 1.4's missing client. The derivation is pure and
unit-tested; only "does the server accept this" needs one running, behind
MESH_TEST_NATS so the ordinary suite stays offline.

A seat's work queue is created at registration, not assignment, so work
queues until a holder appears — a stream created at assignment would make
"the holder is not here yet" mean "your messages are gone". Named after the
seat, because the holder can change and the queued work must not care.

A holder's worker uses a queue group even though the seat guarantees one
holder: the seat is authority, the queue group is delivery, and tying them
together means the day somebody allows two holders every message is
processed twice with nothing reporting it.

One consumer per module carrying every filter, because its ack permission is
derived from its name.

And a real bug the live server caught: a durable name may not contain a dot,
but an ack subject is $JS.ACK.<stream>.<consumer>, so the single string that
read correctly inside the permission was rejected as a consumer name. Split
in two, beside the permission that has to match. Unfixed, the symptom would
have been every message redelivered forever with a permission list that
looks right — which is the failure design 25 §4 warns about.
This commit is contained in:
2026-09-26 22:28:42 +02:00
parent 7232d6df4b
commit aa74bd86ca
7 changed files with 569 additions and 19 deletions
+142
View File
@@ -0,0 +1,142 @@
package broker
import (
"errors"
"fmt"
"time"
"github.com/nats-io/nats.go"
)
// The JetStream side of the controller: the one place the mesh's streams and consumers are
// actually created.
//
// Everything that decides *what* they are is pure and lives beside this (streams.go, derived.go).
// This is only the part that talks to a server, kept small on purpose: a bug in a subject filter
// should be findable in a unit test, and only a bug in "did the server accept it" should need one
// running.
// A JetStream is a connection to the bus, as the controller uses it.
type JetStream struct {
conn *nats.Conn
js nats.JetStreamContext
}
// Dial connects and returns the controller's JetStream handle.
func Dial(url string, opts ...nats.Option) (*JetStream, error) {
// A name, because a connection nobody can identify in the server's own monitoring is one
// nobody can attribute a problem to.
opts = append(opts, nats.Name("mesh-controller"), nats.Timeout(10*time.Second))
conn, err := nats.Connect(url, opts...)
if err != nil {
return nil, fmt.Errorf("connecting to the bus at %s: %w", url, err)
}
js, err := conn.JetStream()
if err != nil {
conn.Close()
return nil, fmt.Errorf("the bus at %s has no JetStream: %w", url, err)
}
return &JetStream{conn: conn, js: js}, nil
}
func (j *JetStream) Close() {
if j.conn != nil {
j.conn.Close()
}
}
// EnsureStream creates the stream if it is absent and brings it to match if it is present.
//
// **Idempotent, because the controller asserts on every start** rather than creating once at
// genesis: a stream somebody deleted, or a mesh raised from a restored backup, has to converge
// rather than run without the guarantee its messages assume.
//
// An update, not a delete and recreate. Recreating would discard every message the stream holds
// and every consumer's position in it — which for CONTROL means the pushes being held through a
// store restart, exactly the guarantee the stream exists for.
func (j *JetStream) EnsureStream(s Stream) error {
want := &nats.StreamConfig{
Name: s.Name,
Subjects: s.Subjects,
Retention: retentionOf(s.Retention),
MaxAge: time.Duration(s.MaxAge) * time.Second,
MaxMsgsPerSubject: int64(s.MaxMsgsPerSubject),
Description: s.Why,
}
if s.Retention == RetentionLastPerSubject {
// Last-per-subject is a limits stream with one message kept per subject, not a
// retention policy of its own — the state shape, spelled the way the server spells it.
want.Retention = nats.LimitsPolicy
want.MaxMsgsPerSubject = 1
want.MaxAge = 0
}
switch _, err := j.js.StreamInfo(s.Name); {
case err == nil:
if _, err := j.js.UpdateStream(want); err != nil {
return fmt.Errorf("bringing stream %s to match: %w", s.Name, err)
}
return nil
case errors.Is(err, nats.ErrStreamNotFound):
if _, err := j.js.AddStream(want); err != nil {
return fmt.Errorf("creating stream %s: %w", s.Name, err)
}
return nil
default:
return fmt.Errorf("asking about stream %s: %w", s.Name, err)
}
}
// EnsureConsumer creates or updates one durable consumer.
//
// Explicit acknowledgement throughout: a consumer that acknowledges on delivery cannot redeliver
// work its holder died in the middle of, which is the whole difference between a queue and a
// firehose.
func (j *JetStream) EnsureConsumer(c Consumer) error {
want := &nats.ConsumerConfig{
Durable: c.Name,
AckPolicy: nats.AckExplicitPolicy,
AckWait: time.Duration(c.AckWaitSeconds) * time.Second,
MaxDeliver: c.MaxDeliver,
DeliverGroup: c.Queue,
DeliverSubject: "",
Description: c.Why,
}
switch len(c.Filters) {
case 0:
case 1:
want.FilterSubject = c.Filters[0]
default:
want.FilterSubjects = c.Filters
}
// A queue group needs a delivery subject: a pull consumer has no group, and declaring one
// without the other is refused by the server with a message that does not say which half is
// missing.
if c.Queue != "" {
want.DeliverSubject = "_DELIVER." + c.Name
}
switch _, err := j.js.ConsumerInfo(c.Stream, c.Name); {
case err == nil:
if _, err := j.js.UpdateConsumer(c.Stream, want); err != nil {
return fmt.Errorf("bringing consumer %s on %s to match: %w", c.Name, c.Stream, err)
}
return nil
case errors.Is(err, nats.ErrConsumerNotFound):
if _, err := j.js.AddConsumer(c.Stream, want); err != nil {
return fmt.Errorf("creating consumer %s on %s: %w", c.Name, c.Stream, err)
}
return nil
default:
return fmt.Errorf("asking about consumer %s on %s: %w", c.Name, c.Stream, err)
}
}
func retentionOf(r Retention) nats.RetentionPolicy {
switch r {
case RetentionWorkQueue:
return nats.WorkQueuePolicy
default:
return nats.LimitsPolicy
}
}