Files
mesh-controller/internal/link/store_window_test.go
T
jschoubben 06cf3c04e5 The consume side behind a seam, and the window wiring into the loop
The outbound half went behind `Bus` and the transport stopped reaching its
callers; this is the other half, and the larger one. Every handler took
`amqp.Delivery`, so the serving loop could not move to another bus without
moving enrolment, reports, builds, upgrades and catch-up with it in one breath.

`Control` states one message in the mesh's words — took it, dropped it, or held
it for the store — and `Inbound` is where messages come from. The AMQP
implementation is today's loop moved rather than changed: same queues, same
prefetch, same holding, because the mesh is running on it and a bus nothing
speaks yet is no reason to alter the one every node is on.

The window (window.go) is now what decides, instead of the conditions that were
inlined in the loop. Two things that surfaced in the wiring:

**Supersession is asked before the store, not after.** A report about a
declaration the mesh has moved past would otherwise wait out a restarting store
to be written and then overwrite what the node is doing now.

**Half of a report is not about a declaration, and that half is never stale.**
What the machine *is* — the tunnel it took over, the ports its own bundle
holds, what an adopted node found, a node moving its overlay key — reaches the
mesh on a report and nowhere else. A rekey set aside as stale is a node whose
overlay key never moves, and no retry is coming, because the node said it once.
So staleness is asked only of a report that is purely an apply's account.

The one thing holding-in-memory can do that holding-in-the-server cannot is
named rather than hidden: `About` sets aside a held message when a newer one
about the same thing arrives, and the bus being built ignores it because the
digest answers the same question.
2026-09-27 00:44:16 +02:00

155 lines
5.8 KiB
Go

package link
import (
"context"
"errors"
"fmt"
"testing"
"github.com/jackc/pgx/v5/pgconn"
)
// What a store restarting under an adoption answers with (novox/hq issues 082, 083).
var restarting = &pgconn.PgError{Code: "57P03", Message: "the database system is starting up"}
type recordsWith struct{ err error }
func (r recordsWith) Built(context.Context, BuildResult) error { return r.err }
type upgradesWith struct{ err error }
func (u upgradesWith) Upgraded(context.Context, Upgraded) error { return u.err }
type replaysWith struct{ err error }
func (r replaysWith) Announceable(context.Context) ([]Announcement, error) { return nil, r.err }
// A build result the store could not take right now is handed back; one it refused is rejected,
// as before; one it kept is acknowledged.
func TestABuildResultWaitsOutARestartingStore(t *testing.T) {
built := BuildResult{On: "anchor", Repository: "/r", Commit: "abc"}
for _, c := range []struct {
what string
err error
want func(*settled) bool
}{
{"restarting", restarting, func(s *settled) bool { return s.unsettled() }},
{"refused", errors.New("no such module"), func(s *settled) bool { return s.rejected && !s.nacked }},
{"kept", nil, func(s *settled) bool { return s.acked && !s.nacked }},
} {
s, in := serving()
s.recorder = recordsWith{err: c.err}
to := &settled{}
s.act(context.Background(), in.sends(t, to, KindBuilt, built))
if !c.want(to) {
t.Errorf("%s: a build result was settled as %+v", c.what, *to)
}
}
}
// An upgrade announcement arriving while the store restarts is asked again; any other failure is
// acknowledged, so it cannot stop every upgrade behind it.
func TestAnUpgradeWaitsOutARestartingStoreAndNothingElse(t *testing.T) {
moved := Upgraded{Module: "gitea", Commit: "abcdef0123"}
for _, c := range []struct {
what string
err error
want func(*settled) bool
}{
{"the store away, said by the upgrader", errors.Join(ErrTryAgain, restarting), func(s *settled) bool { return s.unsettled() }},
{"a push that timed out", context.DeadlineExceeded, func(s *settled) bool { return s.acked && !s.nacked }},
{"cannot act", errors.New("anchor cannot be resolved"), func(s *settled) bool { return s.acked && !s.nacked }},
{"acted", nil, func(s *settled) bool { return s.acked && !s.nacked }},
} {
s, in := serving()
s.upgrader = upgradesWith{err: c.err}
to := &settled{}
s.act(context.Background(), in.sends(t, to, KindModuleMoved, moved))
if !c.want(to) {
t.Errorf("%s: an upgrade was settled as %+v", c.what, *to)
}
}
}
// A catalogue's request to catch up is acknowledged after the work, and asked again while the
// store cannot be read — not lost until the catalogue next restarts.
func TestACatchUpWaitsOutARestartingStore(t *testing.T) {
for _, c := range []struct {
what string
err error
want func(*settled) bool
}{
{"restarting", restarting, func(s *settled) bool { return s.unsettled() }},
{"unreadable", errors.New("a build row is malformed"), func(s *settled) bool { return s.acked && !s.nacked }},
{"nothing to replay", nil, func(s *settled) bool { return s.acked && !s.nacked }},
} {
s, in := serving()
s.replayer = replaysWith{err: c.err}
to := &settled{}
s.act(context.Background(), in.sends(t, to, KindCatchUp, map[string]string{}))
if !c.want(to) {
t.Errorf("%s: a catch-up request was settled as %+v", c.what, *to)
}
}
}
// Shutting down is not an answer about a message: one handled with a cancelled context is left
// unsettled, for the bus to hand to whatever consumes next (issue 083, review).
func TestAMessageHandledDuringShutdownIsLeftForTheBus(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel()
s, in := serving()
s.recorder = recordsWith{err: context.Canceled}
to := &settled{}
s.act(ctx, in.sends(t, to, KindBuilt, BuildResult{On: "anchor", Repository: "/r", Commit: "abc"}))
if !to.unsettled() {
t.Fatalf("a build result handled during shutdown was settled, and so lost: %+v", *to)
}
}
// Two identical build results: the newer sets the older aside rather than leaving it unsettled
// for ever, holding a place in the prefetch.
func TestAnIdenticalBuildResultSetsTheHeldOneAside(t *testing.T) {
s, in := serving()
s.recorder = recordsWith{err: restarting}
built := BuildResult{On: "anchor", Repository: "/r", Commit: "abc"}
first, second := &settled{}, &settled{}
s.act(context.Background(), in.sends(t, first, KindBuilt, built))
s.act(context.Background(), in.sends(t, second, KindBuilt, built))
if !first.acked || !second.unsettled() || len(in.held) != 1 {
t.Fatalf("an identical build result did not set the held one aside: first %+v second %+v, %d held",
*first, *second, len(in.held))
}
}
// What is held stops short of the prefetch, so the loop always has room to answer an enrolment.
func TestWhatIsHeldLeavesRoomInThePrefetch(t *testing.T) {
s, in := serving()
s.recorder = recordsWith{err: restarting}
var last *settled
for i := 0; i < Prefetch; i++ {
last = &settled{}
s.act(context.Background(), in.sends(t, last, KindBuilt, BuildResult{On: "anchor", Commit: fmt.Sprint(i)}))
}
if len(in.held) != Prefetch-PrefetchHeadroom {
t.Fatalf("%d messages were held; the ceiling is %d", len(in.held), Prefetch-PrefetchHeadroom)
}
if last.unsettled() {
t.Fatalf("a message past the ceiling was held: %+v", *last)
}
}
// An upgrade handled during shutdown is left for the bus too — the upgrader's error is the
// cancelled context, which is no answer about the announcement.
func TestAnUpgradeHandledDuringShutdownIsLeftForTheBus(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel()
s, in := serving()
s.upgrader = upgradesWith{err: context.Canceled}
to := &settled{}
s.act(ctx, in.sends(t, to, KindModuleMoved, Upgraded{Module: "gitea", Commit: "abcdef0123"}))
if !to.unsettled() {
t.Fatalf("an upgrade was settled during shutdown, and so lost: %+v", *to)
}
}