A catalogue asks for what it was not there to hear

Its event queue is durable, so a running catalogue misses nothing. What it cannot
have is what was announced before it first ran — and on a fresh mesh that is never
arbitrary: the shared base, the store the catalogue runs on, and the catalogue
itself are each necessarily built BEFORE a catalogue exists to hear about them.
The graph's foundation is the part it never sees.

So it says it is catching up, and the control plane re-announces what it
recorded, oldest first, marked as a replay. Oldest first because a graph is built
in the order things happened: registering a module that stands on a base before
the base would point an edge at a version nothing has seen, and the shape of a
fresh mesh guarantees the base is both first and the one that was missed.

The replayer hands announcements back rather than publishing them, because the
wire belongs to the link package and a replay building its own events could drift
from what the builder emits — the one thing it must match exactly, since the
catalogue has a single handler for both.

Its own queue and its own consumer: two consumers on one queue split its
messages, and a catch-up request going to whichever half was not listening is a
gap that looks like a working mesh.

Toward novox/hq 04-ISSUES/050.

Claude-Session: https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx
This commit is contained in:
2026-09-15 01:26:58 +02:00
parent 2b82872ac3
commit 3ae7b88c6d
5 changed files with 214 additions and 0 deletions
+67
View File
@@ -51,6 +51,7 @@ type Server struct {
recorder Recorder
log *log.Logger
upgrader Upgrader
replayer Replayer
}
// Records tells the server where to keep build results.
@@ -89,6 +90,21 @@ func (s *Server) Follows(u Upgrader) error {
return nil
}
// Answers binds the queue a catalogue's catch-up request arrives on.
//
// **Not bound unless something is listening**, for the same reason upgrades are not: a durable
// queue with no consumer fills quietly and the first symptom is a broker out of disk.
func (s *Server) Answers(r Replayer) error {
if _, err := s.channel.QueueDeclare(CatchUpQueue, true, false, false, false, nil); err != nil {
return fmt.Errorf("cannot declare the %s queue: %w", CatchUpQueue, err)
}
if err := s.channel.QueueBind(CatchUpQueue, KeyCatchingUp, EventsExchange, false, nil); err != nil {
return fmt.Errorf("cannot bind %s to %s/%s: %w", CatchUpQueue, EventsExchange, KeyCatchingUp, err)
}
s.replayer = r
return nil
}
// Connect opens the control plane's own connection to the broker.
func Connect(enroller Enroller, listener Listener) (*Server, error) {
url := strings.TrimSpace(os.Getenv(AMQPVar))
@@ -187,6 +203,18 @@ func (s *Server) Serve(ctx context.Context) error {
}
}
// Its own queue and its own consumer, for the reason above: two consumers on one queue split
// its messages, and a catch-up request going to whichever half was not listening is a gap that
// looks like a working mesh.
var catchups <-chan amqp.Delivery
if s.replayer != nil {
catchups, err = s.channel.ConsumeWithContext(ctx, CatchUpQueue, "control-plane-catchup",
false, false, false, false, nil)
if err != nil {
return err
}
}
closed := s.conn.NotifyClose(make(chan *amqp.Error, 1))
s.log.Printf("consuming %s, bound to %s/{%s,%s,%s,%s}",
ControlQueue, Exchange, KeyEnrol, KeyReport, KeyAlive, KeyBuilt)
@@ -198,6 +226,14 @@ func (s *Server) Serve(ctx context.Context) error {
select {
case <-ctx.Done():
return nil
case delivery, ok := <-catchups:
if !ok {
if catchups != nil {
return errors.New("the broker stopped delivering catch-up requests")
}
continue
}
s.catchingUp(ctx, delivery)
case delivery, ok := <-upgrades:
// A nil channel blocks for ever, so this case simply never fires when nothing is
// listening for upgrades. Closed is different, and means the broker stopped.
@@ -379,6 +415,37 @@ func (s *Server) handleBuilt(ctx context.Context, delivery amqp.Delivery) {
// none of those get better by being handed the same message again. Requeuing would put a poison
// message at the head of a durable queue and stop every upgrade behind it, which turns one module
// nobody can push into a mesh that stops following its own catalogue.
// catchingUp answers a catalogue that has just started and may have missed builds.
//
// Acknowledged before the work, deliberately: a replay that fails is not one that succeeds by
// being handed the same request again, and the catalogue asks every time it starts. Requeueing a
// poison request would stop every later catch-up behind it.
func (s *Server) catchingUp(ctx context.Context, delivery amqp.Delivery) {
defer func() { _ = delivery.Ack(false) }()
if s.replayer == nil {
s.log.Printf("a catalogue asked to catch up and this control plane has nothing to replay")
return
}
announcements, err := s.replayer.Announceable(ctx)
if err != nil {
s.log.Printf("a catalogue asked to catch up and the mesh could not read its builds: %v", err)
return
}
sent := 0
for _, a := range announcements {
a.Replay = true
if err := EmitEvent(ctx, s.channel, KeyModuleBuilt, "control-plane", "", a); err != nil {
// Said and abandoned rather than retried: the catalogue asks again every time it
// starts, and half a graph delivered twice is no better than half delivered once.
s.log.Printf("replaying %s at %s failed, and the rest is abandoned: %v",
a.Module, short(a.Commit), err)
return
}
sent++
}
s.log.Printf("a catalogue asked to catch up; re-announced %d build(s)", sent)
}
func (s *Server) upgraded(ctx context.Context, delivery amqp.Delivery) {
defer func() { _ = delivery.Ack(false) }()
var u Upgraded