Review of 083: a message the store cannot take is held and retried on a ticker, not slept on, so enrolments are answered meanwhile; a newer one per subject supersedes; the password is replaced after the spend; the same presenter may finish after a lost answer; upgrades retry only on the store

This commit is contained in:
2026-09-22 14:23:03 +02:00
parent 1a41b88ed3
commit a3b7e830c8
7 changed files with 280 additions and 184 deletions
+13 -2
View File
@@ -28,13 +28,15 @@ type following struct{ open *stores }
func (f following) Upgraded(ctx context.Context, u link.Upgraded) error { func (f following) Upgraded(ctx context.Context, u link.Upgraded) error {
inv := f.open.inventory inv := f.open.inventory
// The store read first, and an outage there said as one, so the announcement is held and asked
// again (novox/hq issue 083). Only here: a push that fails further down is not asked again.
decision, err := inv.UpgradeOf(ctx, u.Module) decision, err := inv.UpgradeOf(ctx, u.Module)
if err != nil { if err != nil {
return err return notNow(err)
} }
on, err := inv.Running(ctx, u.Module) on, err := inv.Running(ctx, u.Module)
if err != nil { if err != nil {
return err return notNow(err)
} }
if len(on) == 0 { if len(on) == 0 {
fmt.Printf("%s moved to %s; no machine runs it\n", u.Module, shortCommit(u.Commit)) fmt.Printf("%s moved to %s; no machine runs it\n", u.Module, shortCommit(u.Commit))
@@ -194,3 +196,12 @@ func (f following) Announceable(ctx context.Context) ([]link.Announcement, error
} }
return out, nil return out, nil
} }
// notNow marks a store that could not be read right now, so the announcement is held rather than
// lost; anything else is returned as it was.
func notNow(err error) error {
if inventory.Unreachable(err) {
return fmt.Errorf("%w: %w", link.ErrTryAgain, err)
}
return err
}
+12 -4
View File
@@ -413,10 +413,18 @@ func TestATokenIsClaimedByOnePresenterAndSpentOnlyByIt(t *testing.T) {
if err := inv.Spend(ctx, issued.Secret, "key-a"); err != nil { if err := inv.Spend(ctx, issued.Secret, "key-a"); err != nil {
t.Fatalf("the presenter holding the claim could not spend it: %v", err) t.Fatalf("the presenter holding the claim could not spend it: %v", err)
} }
for _, by := range []string{"key-a", "key-b"} { // Spent: nobody else may claim it, ever.
if _, err := inv.Claim(ctx, issued.Secret, by); !errors.Is(err, ErrTokenRefused) { if _, err := inv.Claim(ctx, issued.Secret, "key-b"); !errors.Is(err, ErrTokenRefused) {
t.Fatalf("a spent token was claimed again by %s: %v", by, err) t.Fatalf("a spent token was claimed by another presenter: %v", err)
} }
// But the presenter that spent it may, and spend it again: its spend reached the store and the
// answer did not reach the node, which asked again — refusing it would lock out a machine the
// mesh holds as enrolled.
if _, err := inv.Claim(ctx, issued.Secret, "key-a"); err != nil {
t.Fatalf("the presenter whose answer was lost after its spend was refused: %v", err)
}
if err := inv.Spend(ctx, issued.Secret, "key-a"); err != nil {
t.Fatalf("spending again by the same presenter failed: %v", err)
} }
} }
+27 -16
View File
@@ -203,15 +203,6 @@ func (i *Inventory) IssueToken(ctx context.Context, nodeName string, validFor ti
// token that had expired. // token that had expired.
var ErrTokenRefused = errors.New("that token cannot be used") var ErrTokenRefused = errors.New("that token cannot be used")
// Redeem spends a token and reports which node it was for.
//
// It does not issue an identity. What a node presents afterwards to prove it is that node is not
// decided anywhere (novox/hq ADR 0004 names the property, not the mechanism), and guessing at it
// in a migration is the most expensive guess available here.
//
// The update is the check: one statement that both finds a live token and marks it used, so two
// simultaneous redemptions of one secret cannot both succeed. Reading first and writing second
// would leave exactly that gap.
// ClaimLease is how long a claimed token is held for the one presenter that claimed it. Long // ClaimLease is how long a claimed token is held for the one presenter that claimed it. Long
// enough for an enrolment to be tried again through a store restart; short enough that a host // enough for an enrolment to be tried again through a store restart; short enough that a host
// which gave up and was started over, with keys of its own, is not kept waiting long. // which gave up and was started over, with keys of its own, is not kept waiting long.
@@ -224,12 +215,17 @@ var ErrTokenInUse = errors.New("the token is being used by another enrolment")
// Claim takes a token for one presenter — `by`, which names the key presenting it — for the length // Claim takes a token for one presenter — `by`, which names the key presenting it — for the length
// of a lease, and says which node it enrols. The same presenter may claim it again, as may anyone // of a lease, and says which node it enrols. The same presenter may claim it again, as may anyone
// once the lease has lapsed; nothing is spent until Spend (novox/hq 04-ISSUES/083). // once the lease has lapsed; nothing is spent until Spend (novox/hq 04-ISSUES/083).
//
// A token this same presenter already spent is claimed again too: its spend reached the store and
// the answer did not reach the node, which asked again. Refusing it then would lock out a machine
// the mesh holds as enrolled — with the key it is still presenting.
func (i *Inventory) Claim(ctx context.Context, secret, by string) (Node, error) { func (i *Inventory) Claim(ctx context.Context, secret, by string) (Node, error) {
var id string var id string
err := i.store.Pool().QueryRow(ctx, err := i.store.Pool().QueryRow(ctx,
`update enrolment_token set claimed_by = $2, claimed_until = now() + $3::interval `update enrolment_token set claimed_by = $2, claimed_until = now() + $3::interval
where secret = $1 and redeemed is null and expires > now() where secret = $1 and expires > now()
and (claimed_by is null or claimed_by = $2 or claimed_until < now()) and ((redeemed is null and (claimed_by is null or claimed_by = $2 or claimed_until < now()))
or (redeemed is not null and claimed_by = $2))
returning node`, hashSecret(secret), by, ClaimLease.String()).Scan(&id) returning node`, hashSecret(secret), by, ClaimLease.String()).Scan(&id)
if errors.Is(err, pgx.ErrNoRows) { if errors.Is(err, pgx.ErrNoRows) {
// Unusable, or held by someone else — told apart, because the second passes. // Unusable, or held by someone else — told apart, because the second passes.
@@ -237,8 +233,12 @@ func (i *Inventory) Claim(ctx context.Context, secret, by string) (Node, error)
probe := i.store.Pool().QueryRow(ctx, probe := i.store.Pool().QueryRow(ctx,
`select true from enrolment_token `select true from enrolment_token
where secret = $1 and redeemed is null and expires > now()`, hashSecret(secret)).Scan(&held) where secret = $1 and redeemed is null and expires > now()`, hashSecret(secret)).Scan(&held)
if probe == nil && held { switch {
case probe == nil && held:
return Node{}, ErrTokenInUse return Node{}, ErrTokenInUse
case probe != nil && !errors.Is(probe, pgx.ErrNoRows):
// The store went away between the two questions: "not now", not a refusal.
return Node{}, probe
} }
return Node{}, ErrTokenRefused return Node{}, ErrTokenRefused
} }
@@ -251,12 +251,13 @@ func (i *Inventory) Claim(ctx context.Context, secret, by string) (Node, error)
return n, err return n, err
} }
// Spend makes a claimed token used, only for the presenter holding the claim. The last write of an // Spend makes a claimed token used, only for the presenter holding the claim. The last write to the
// enrolment, so a token is spent exactly when the node it enrolled is complete. // store in an enrolment, so a token is spent exactly when the node it enrolled is complete. Spent
// again by the same presenter is not an error: an answer lost after the first spend.
func (i *Inventory) Spend(ctx context.Context, secret, by string) error { func (i *Inventory) Spend(ctx context.Context, secret, by string) error {
tag, err := i.store.Pool().Exec(ctx, tag, err := i.store.Pool().Exec(ctx,
`update enrolment_token set redeemed = now() `update enrolment_token set redeemed = coalesce(redeemed, now())
where secret = $1 and redeemed is null and claimed_by = $2`, hashSecret(secret), by) where secret = $1 and claimed_by = $2`, hashSecret(secret), by)
if err != nil { if err != nil {
return err return err
} }
@@ -266,6 +267,16 @@ func (i *Inventory) Spend(ctx context.Context, secret, by string) error {
return nil return nil
} }
// Redeem spends a token in one step and reports which node it was for. Enrolment claims and then
// spends (Claim, Spend); this is the one-step form, kept for what spends a token outright.
//
// It does not issue an identity. What a node presents afterwards to prove it is that node is not
// decided anywhere (novox/hq ADR 0004 names the property, not the mechanism), and guessing at it
// in a migration is the most expensive guess available here.
//
// The update is the check: one statement that both finds a live token and marks it used, so two
// simultaneous redemptions of one secret cannot both succeed. Reading first and writing second
// would leave exactly that gap.
func (i *Inventory) Redeem(ctx context.Context, secret string) (Node, error) { func (i *Inventory) Redeem(ctx context.Context, secret string) (Node, error) {
var id string var id string
err := i.store.Pool().QueryRow(ctx, err := i.store.Pool().QueryRow(ctx,
+28 -21
View File
@@ -30,12 +30,15 @@ type Enrolment struct {
Broker broker.Broker Broker broker.Broker
} }
// Enrol spends the token and records what the node presented. // Enrol records what the node presented and spends the token.
// //
// Order matters and it is the order things become irreversible. The token is spent first, in a // Order matters and it is the order things become irreversible (novox/hq issue 083). The token is
// single statement that both finds and marks it, so two machines racing on one secret produce one // claimed first, in a single statement that both finds it and holds it for this presenter's key, so
// winner. Only then is a key recorded — because recording a key for a node whose token turned out // two machines racing on one secret produce one holder. Then everything the node presented is
// to be spent would leave the mesh believing a machine that never had the right to join. // written — each write an overwrite, so an attempt interrupted by the store going away can be made
// again by the same presenter. Then the token is spent. Last, the token's secret stops being the
// node's broker password: done after the spend, because a password replaced by an attempt that
// then failed would be one nobody holds, and the node could not even log in to ask again.
func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (reply EnrolReply, err error) { func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (reply EnrolReply, err error) {
secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile secret, public, profile := request.Secret, ed25519.PublicKey(request.PublicKey), request.Profile
@@ -81,21 +84,6 @@ func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (reply Enrol
Signer: key.Public, Signer: key.Public,
} }
// The token's secret was the broker password up to this moment, which is what let this
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
// credential the node keeps for years is not the one that was pasted into a terminal.
if e.Management != nil {
password, err := freshPassword()
if err != nil {
return EnrolReply{}, err
}
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
return EnrolReply{}, fmt.Errorf(
"%s's broker password could not be replaced: %w", node.Name, err)
}
reply.Password = password
}
// Recorded before the profile because the overlay is the first declaration this node will // Recorded before the profile because the overlay is the first declaration this node will
// receive, and without this key the mesh cannot compose one. A node enrolled with no overlay // receive, and without this key the mesh cannot compose one. A node enrolled with no overlay
// key is a node the graph skips — an ordinary in-between state, and one worth leaving as // key is a node the graph skips — an ordinary in-between state, and one worth leaving as
@@ -124,11 +112,30 @@ func (e Enrolment) Enrol(ctx context.Context, request EnrolRequest) (reply Enrol
} }
} }
// Spent last, so a token is used exactly when the node it enrolled is complete. // Spent once the node is complete in the store.
if err := e.Inventory.Spend(ctx, secret, by); err != nil { if err := e.Inventory.Spend(ctx, secret, by); err != nil {
return EnrolReply{}, fmt.Errorf("%s was written and its token could not be spent: %w", node.Name, err) return EnrolReply{}, fmt.Errorf("%s was written and its token could not be spent: %w", node.Name, err)
} }
// The token's secret was the broker password up to this moment, which is what let this
// connection exist at all. It is replaced now, so the one-time thing stays one-time and the
// credential the node keeps for years is not the one that was pasted into a terminal. After
// the spend and not before: a replaced password on an attempt that failed would be held by
// nobody. If the broker will not take it now, the enrolment still stands — the node keeps
// the token's secret as its password, which it is told, and which is said here.
if e.Management != nil {
password, err := freshPassword()
if err != nil {
return EnrolReply{}, err
}
if err := e.Management.CreateNodeAccount(ctx, node.Name, password); err != nil {
log.Printf("%s is enrolled and its broker password could not be replaced, so it keeps "+
"the token's secret as its password: %v", node.Name, err)
} else {
reply.Password = password
}
}
if profile != nil { if profile != nil {
// Not fatal if it fails. The profile is what the control plane needs in order to decide // Not fatal if it fails. The profile is what the control plane needs in order to decide
// what this machine should run, and it is reported again on every connection — so losing // what this machine should run, and it is reported again on every connection — so losing
+54 -54
View File
@@ -20,85 +20,85 @@ func (a *saidTo) Nack(_ uint64, _ bool, requeue bool) error {
return nil return nil
} }
func (a *saidTo) Reject(uint64, bool) error { return nil } func (a *saidTo) Reject(uint64, bool) error { return nil }
func (a *saidTo) unsettled() bool { return !a.acked && !a.nacked }
type heardWith struct{ err error } type heardWith struct{ err error }
func (h heardWith) Heard(context.Context, Report) error { return h.err } func (h heardWith) Heard(context.Context, Report) error { return h.err }
func aReport(t *testing.T, to *saidTo) amqp.Delivery { // switchable answers with whatever it is set to — the store away, then back.
type switchable struct{ err error }
func (h *switchable) Heard(context.Context, Report) error { return h.err }
var tag uint64
func aReport(t *testing.T, to *saidTo, node, declared string) amqp.Delivery {
t.Helper() t.Helper()
body, err := json.Marshal(Report{Node: "anchor", Applied: []string{"store"}}) body, err := json.Marshal(Report{Node: node, Declared: declared, Applied: []string{"store"}})
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
return amqp.Delivery{Acknowledger: to, RoutingKey: KeyReport, Body: body} tag++
return amqp.Delivery{Acknowledger: to, RoutingKey: KeyReport, Body: body, DeliveryTag: tag}
} }
// A report the store could not take right now goes back to the broker to be asked again; one the func quiet() *log.Logger { return log.New(io.Discard, "", 0) }
// store answered no to is acknowledged, or it would come back for ever (novox/hq issue 082).
func TestAReportTheStoreCouldNotTakeIsKeptAndOneItRefusedIsNot(t *testing.T) {
quiet := log.New(io.Discard, "", 0)
notNow := &saidTo{} // A report the store could not take right now is held, unsettled, and recorded when the store is
s := &Server{listener: heardWith{err: errors.Join(ErrTryAgain, errors.New("starting up"))}, log: quiet, again: 1} // back; one the store answered no to is acknowledged; one recorded is acknowledged (issue 082, 083).
s.handleReport(context.Background(), aReport(t, notNow)) func TestAReportTheStoreCouldNotTakeIsHeldAndOneItRefusedIsNot(t *testing.T) {
if notNow.acked || !notNow.nacked || !notNow.requeued { store := &switchable{err: errors.Join(ErrTryAgain, errors.New("starting up"))}
t.Fatalf("a report the store could not take yet was not handed back to be asked again: %+v", notNow) s := &Server{listener: store, log: quiet()}
held := &saidTo{}
s.handleReport(context.Background(), aReport(t, held, "anchor", "d1"))
if !held.unsettled() || len(s.parked) != 1 {
t.Fatalf("a report the store could not take was not held: %+v, %d held", held, len(s.parked))
}
store.err = nil
s.retryHeld(context.Background())
if !held.acked || len(s.parked) != 0 {
t.Fatalf("a held report was not recorded once the store was back: %+v, %d held", held, len(s.parked))
} }
refused := &saidTo{} refused := &saidTo{}
s = &Server{listener: heardWith{err: errors.New("a report named no node")}, log: quiet, again: 1} s = &Server{listener: heardWith{err: errors.New("a report named no node")}, log: quiet()}
s.handleReport(context.Background(), aReport(t, refused)) s.handleReport(context.Background(), aReport(t, refused, "anchor", "d1"))
if !refused.acked || refused.nacked { if !refused.acked || refused.nacked {
t.Fatalf("a report the store answered no to was not acknowledged, so it would spin: %+v", refused) t.Fatalf("a report the store answered no to was not acknowledged: %+v", refused)
} }
}
recorded := &saidTo{} // A newer report from the same node supersedes one of its reports still held: recorded after the
s = &Server{listener: heardWith{}, log: quiet} // newer, the older would overwrite what the node is doing now.
s.handleReport(context.Background(), aReport(t, recorded)) func TestANewerReportSupersedesAHeldOneFromTheSameNode(t *testing.T) {
if !recorded.acked || recorded.nacked { s := &Server{listener: heardWith{err: errors.Join(ErrTryAgain, errors.New("starting up"))}, log: quiet()}
t.Fatalf("a recorded report was not acknowledged: %+v", recorded) older, newer, other := &saidTo{}, &saidTo{}, &saidTo{}
s.handleReport(context.Background(), aReport(t, older, "anchor", "d1"))
s.handleReport(context.Background(), aReport(t, other, "laptop", "d7"))
s.handleReport(context.Background(), aReport(t, newer, "anchor", "d2"))
if !older.acked {
t.Fatalf("the older report was not set aside by the newer: %+v", older)
}
if !newer.unsettled() || !other.unsettled() || len(s.parked) != 2 {
t.Fatalf("the newer report and another node's were not both held: newer %+v other %+v, %d held",
newer, other, len(s.parked))
} }
} }
// A store that has not come back within the bound is not restarting: the report is let go, loudly, // A store that has not come back within the bound is not restarting: the report is let go, loudly,
// rather than holding every enrolment and report behind it for ever (novox/hq issue 082, review). // rather than held for ever.
func TestAReportIsLetGoOnceTheStoreHasBeenGoneTooLong(t *testing.T) { func TestAReportIsLetGoOnceTheStoreHasBeenGoneTooLong(t *testing.T) {
quiet := log.New(io.Discard, "", 0)
s := &Server{listener: heardWith{err: errors.Join(ErrTryAgain, errors.New("connection refused"))}, s := &Server{listener: heardWith{err: errors.Join(ErrTryAgain, errors.New("connection refused"))},
log: quiet, again: 1, giveUp: time.Millisecond} log: quiet(), giveUp: time.Millisecond}
held := &saidTo{}
first := &saidTo{} s.handleReport(context.Background(), aReport(t, held, "anchor", "d1"))
s.handleReport(context.Background(), aReport(t, first)) if !held.unsettled() {
if !first.requeued { t.Fatalf("the first failure was not held: %+v", held)
t.Fatalf("the first failure was not handed back: %+v", first)
} }
time.Sleep(5 * time.Millisecond) time.Sleep(5 * time.Millisecond)
later := &saidTo{} s.retryHeld(context.Background())
s.handleReport(context.Background(), aReport(t, later)) if !held.acked || len(s.parked) != 0 {
if !later.acked || later.nacked { t.Fatalf("a report past the bound was not let go: %+v, %d held", held, len(s.parked))
t.Fatalf("a report past the bound was not let go, so it would hold the queue for ever: %+v", later)
}
// Let go, and forgotten: the same report arriving fresh starts a new clock.
again := &saidTo{}
s.handleReport(context.Background(), aReport(t, again))
if !again.requeued {
t.Fatalf("a report let go was not forgotten, so its next arrival is given no chance: %+v", again)
}
}
// Shutting down does not wait out the pause.
func TestAReportBeingTriedAgainDoesNotHoldUpShutdown(t *testing.T) {
quiet := log.New(io.Discard, "", 0)
s := &Server{listener: heardWith{err: errors.Join(ErrTryAgain, errors.New("starting up"))},
log: quiet, again: time.Hour}
ctx, cancel := context.WithCancel(context.Background())
cancel()
done := make(chan struct{})
go func() { s.handleReport(ctx, aReport(t, &saidTo{})); close(done) }()
select {
case <-done:
case <-time.After(5 * time.Second):
t.Fatal("a cancelled context still waited out the pause")
} }
} }
+137 -82
View File
@@ -55,29 +55,40 @@ type Server struct {
log *log.Logger log *log.Logger
upgrader Upgrader upgrader Upgrader
replayer Replayer replayer Replayer
// again is how long a report the store could not take waits before it is handed back to // Messages the store could not take right now, held unacknowledged and tried again on a
// the broker; zero means TryAgainAfter. giveUp is how long one report is kept trying before // ticker, by subject (novox/hq issues 082, 083). again is the ticker's interval, zero meaning
// it is let go; zero means GiveUpAfter. waiting is when each report still trying first failed. // TryAgainAfter; giveUp is how long one is kept, zero meaning GiveUpAfter.
again time.Duration again time.Duration
giveUp time.Duration giveUp time.Duration
waiting map[string]time.Time parked map[string]*held
}
// held is one message the store could not take, kept to be tried again.
type held struct {
delivery amqp.Delivery
retry func(context.Context, amqp.Delivery)
what string
first time.Time
} }
// ErrTryAgain marks a listener's failure as "not now": what it was given is worth keeping and // ErrTryAgain marks a listener's failure as "not now": what it was given is worth keeping and
// asking again, as when the store is restarting (novox/hq issue 082). // asking again, as when the store is restarting (novox/hq issue 082).
var ErrTryAgain = errors.New("not now, try again") var ErrTryAgain = errors.New("not now, try again")
// TryAgainAfter is the pause before a report the store could not take goes back to the broker. // TryAgainAfter is how often messages the store could not take are tried again. A store comes
// The consumer takes one message at a time, so without it a restarting store would be asked in // back in seconds, and a report a few seconds late is still current.
// a tight loop; a store comes back in seconds, and a report a few seconds late is still current.
const TryAgainAfter = 2 * time.Second const TryAgainAfter = 2 * time.Second
// GiveUpAfter bounds how long one report holds the queue. The consumer takes one message at a // GiveUpAfter bounds how long one message is kept trying. A store that has not come back in this
// time, so a report being tried again holds every enrolment, build result and other report // long is not restarting, and the message is let go with a line saying it was lost.
// behind it; a store that has not come back in this long is not restarting, and holding the
// mesh's control queue for it would turn one lost report into a mesh that answers nothing.
const GiveUpAfter = 2 * time.Minute const GiveUpAfter = 2 * time.Minute
// Prefetch is how many messages the broker hands the control plane before it has settled them.
// More than one because a message the store could not take is held, unsettled, while the loop goes
// on answering others — an enrolment above all, which a host is waiting on (novox/hq issue 083).
// Bounded, because what is held is also what the broker has not kept on its own disk as pending.
const Prefetch = 64
// Records tells the server where to keep build results. // Records tells the server where to keep build results.
// //
// Set after Connect rather than passed to it, because a control plane that only publishes — the // Set after Connect rather than passed to it, because a control plane that only publishes — the
@@ -206,10 +217,11 @@ func (s *Server) Close() {
// receive half of what it expects — a fault this project has already had, between a module's // receive half of what it expects — a fault this project has already had, between a module's
// daemon and its capability server. // daemon and its capability server.
func (s *Server) Serve(ctx context.Context) error { func (s *Server) Serve(ctx context.Context) error {
// Prefetch of one. The control plane writes to a database per message, and a burst of // A bounded prefetch rather than one. The loop still takes messages one at a time; what the
// enrolments delivered all at once would be held in memory rather than left on the broker, // prefetch buys is that a message the store could not take can be held while the loop goes on
// which is the one place they survive a restart. // to the next, instead of every enrolment waiting behind it (novox/hq issue 083). Anything held
if err := s.channel.Qos(1, 0, false); err != nil { // goes back to the broker if the control plane stops, because nothing held is acknowledged.
if err := s.channel.Qos(Prefetch, 0, false); err != nil {
return err return err
} }
@@ -250,10 +262,19 @@ func (s *Server) Serve(ctx context.Context) error {
s.log.Printf("consuming %s, bound to %s/%s", UpgradeQueue, EventsExchange, KeyModuleUpgraded) s.log.Printf("consuming %s, bound to %s/%s", UpgradeQueue, EventsExchange, KeyModuleUpgraded)
} }
again := s.again
if again == 0 {
again = TryAgainAfter
}
ticker := time.NewTicker(again)
defer ticker.Stop()
for { for {
select { select {
case <-ctx.Done(): case <-ctx.Done():
return nil return nil
case <-ticker.C:
s.retryHeld(ctx)
case delivery, ok := <-catchups: case delivery, ok := <-catchups:
if !ok { if !ok {
if catchups != nil { if catchups != nil {
@@ -334,13 +355,17 @@ func (s *Server) handleReport(ctx context.Context, delivery amqp.Delivery) {
_ = delivery.Reject(false) _ = delivery.Reject(false)
return return
} }
// A node's newer report supersedes one of its older reports still held: the older is its
// past, and recorded after the newer it would overwrite what the node is doing now.
subject := "report " + report.Node
s.supersede(subject, delivery)
if s.listener != nil { if s.listener != nil {
if err := s.listener.Heard(context.Background(), report); err != nil { if err := s.listener.Heard(context.Background(), report); err != nil {
// Kept, not acknowledged, while the store cannot take it: the node reports an apply // Held, not acknowledged, while the store cannot take it: the node reports an apply
// once, and a report lost here is a node the mesh never hears from again — the store // once, and a report lost here is a node the mesh never hears from again — the store
// restarting under the adoption that node just applied lost exactly that (issue 082). // restarting under the adoption that node just applied lost exactly that (issue 082).
what := fmt.Sprintf("%s's report of declaration %s", report.Node, report.Declared) what := fmt.Sprintf("%s's report of declaration %s", report.Node, report.Declared)
if s.tryLater(ctx, delivery, what, err) { if s.tryLater(delivery, subject, what, err, s.handle) {
return return
} }
// Said rather than swallowed. A report the mesh heard and failed to write down is a // Said rather than swallowed. A report the mesh heard and failed to write down is a
@@ -357,59 +382,82 @@ func (s *Server) handleReport(ctx context.Context, delivery amqp.Delivery) {
default: default:
s.log.Printf("%s applied %d resource(s)", report.Node, len(report.Applied)) s.log.Printf("%s applied %d resource(s)", report.Node, len(report.Applied))
} }
s.settled(delivery) s.settled(subject, delivery)
_ = delivery.Ack(false) _ = delivery.Ack(false)
} }
// tryLater hands a message the store could not take right now back to the broker, to be asked // tryLater holds a message the store could not take right now, to be tried again on the ticker,
// again after a pause, and says whether it did (novox/hq issues 082, 083). // and says whether it did (novox/hq issues 082, 083).
// //
// "Right now" is the store unreachable or restarting — ErrTryAgain from a listener, or an error // "Right now" is the store unreachable or restarting — ErrTryAgain from a listener, or an error the
// the inventory reads as an outage. Anything else is an answer, and is left to the caller to // inventory reads as an outage. Anything else is an answer, and is left to the caller to settle.
// settle. One message holds its queue at most giveUpAfter: the consumer takes one message at a // Held means unacknowledged and set aside: the loop goes on to the next message, so an enrolment a
// time, so a message tried again holds everything behind it, and a store that has not come back // host is waiting on is answered while a report waits for the store. One message is held at most
// in that long is not restarting. Past it, the message is let go with a line saying it was lost. // giveUpAfter; past it, it is let go with a line saying it was lost, and the caller settles it.
func (s *Server) tryLater(ctx context.Context, delivery amqp.Delivery, what string, err error) bool { func (s *Server) tryLater(delivery amqp.Delivery, subject, what string, err error,
retry func(context.Context, amqp.Delivery)) bool {
if !errors.Is(err, ErrTryAgain) && !inventory.Unreachable(err) { if !errors.Is(err, ErrTryAgain) && !inventory.Unreachable(err) {
return false return false
} }
key := waitingKey(delivery) if s.parked == nil {
if s.waiting == nil { s.parked = map[string]*held{}
s.waiting = map[string]time.Time{}
} }
first, seen := s.waiting[key] h, ok := s.parked[subject]
if !seen { if !ok || h.delivery.DeliveryTag != delivery.DeliveryTag {
first = time.Now() h = &held{delivery: delivery, retry: retry, what: what, first: time.Now()}
s.waiting[key] = first s.parked[subject] = h
s.log.Printf("could not keep %s yet; holding it to try again: %v", what, err)
return true
} }
if time.Since(first) >= s.giveUpAfter() { if time.Since(h.first) >= s.giveUpAfter() {
delete(s.waiting, key) delete(s.parked, subject)
s.log.Printf("LOST %s: the store has not come back in %s, and the queue cannot wait longer: %v", s.log.Printf("LOST %s: the store has not come back in %s: %v", what, s.giveUpAfter(), err)
what, s.giveUpAfter(), err)
return false return false
} }
again := s.again
if again == 0 {
again = TryAgainAfter
}
s.log.Printf("could not keep %s yet, and will again in %s: %v", what, again, err)
select {
case <-ctx.Done():
case <-time.After(again):
}
_ = delivery.Nack(false, true)
return true return true
} }
// settled forgets a message's time spent waiting, once it has been handled either way. // supersede drops a message held for a subject when a newer one for it arrives: the older is
func (s *Server) settled(delivery amqp.Delivery) { delete(s.waiting, waitingKey(delivery)) } // acknowledged, because acting on it after the newer would undo the newer.
func (s *Server) supersede(subject string, newer amqp.Delivery) {
h, ok := s.parked[subject]
if !ok || h.delivery.DeliveryTag == newer.DeliveryTag {
return
}
delete(s.parked, subject)
s.log.Printf("set aside %s: a newer one arrived", h.what)
_ = h.delivery.Ack(false)
}
// waitingKey is a message by its content: the same message handed back is the same key. // settled forgets a message once it has been handled either way.
func waitingKey(delivery amqp.Delivery) string { func (s *Server) settled(subject string, delivery amqp.Delivery) {
sum := sha256.Sum256(append([]byte(delivery.RoutingKey+"\x00"), delivery.Body...)) if h, ok := s.parked[subject]; ok && h.delivery.DeliveryTag == delivery.DeliveryTag {
delete(s.parked, subject)
}
}
// digest names a message by its content.
func digest(body []byte) string {
sum := sha256.Sum256(body)
return hex.EncodeToString(sum[:]) return hex.EncodeToString(sum[:])
} }
// retryHeld tries every held message again. Each handler holds it again, settles it, or lets it
// go past the bound.
func (s *Server) retryHeld(ctx context.Context) {
for _, h := range s.snapshot() {
h.retry(ctx, h.delivery)
}
}
func (s *Server) snapshot() []*held {
out := make([]*held, 0, len(s.parked))
for _, h := range s.parked {
out = append(out, h)
}
return out
}
func (s *Server) giveUpAfter() time.Duration { func (s *Server) giveUpAfter() time.Duration {
if s.giveUp == 0 { if s.giveUp == 0 {
return GiveUpAfter return GiveUpAfter
@@ -445,9 +493,8 @@ func (s *Server) handleEnrol(ctx context.Context, delivery amqp.Delivery) {
s.reply(ctx, delivery, reply) s.reply(ctx, delivery, reply)
// Acknowledged after the reply is sent, so a control plane that dies mid-answer leaves the // Acknowledged after the reply is sent, so a control plane that dies mid-answer leaves the
// request on the broker rather than having consumed it silently. Enrolment is idempotent // request on the broker rather than having consumed it silently. Asked again by the same
// only in the sense that the token is spent — a redelivery gets the refusal, which is // presenter, an enrolment finishes: the token is held for its key and spent last (issue 083).
// correct and visible, where a lost request is neither.
_ = delivery.Ack(false) _ = delivery.Ack(false)
} }
@@ -493,18 +540,20 @@ func (s *Server) handleBuilt(ctx context.Context, delivery amqp.Delivery) {
_ = delivery.Reject(false) _ = delivery.Reject(false)
return return
} }
// Each build result its own subject: none supersedes another, and recording one twice is
// harmless — the build is kept by its id.
subject := "build " + digest(delivery.Body)
if err := s.recorder.Built(ctx, result); err != nil { if err := s.recorder.Built(ctx, result); err != nil {
// Kept while the store cannot take it: a build result lost here is never announced, and // Held while the store cannot take it: a build result lost here is never announced (083).
// recording one twice is harmless — the build is kept by its id (issue 083). if s.tryLater(delivery, subject, fmt.Sprintf("a build result from %s", result.On), err, s.handle) {
if s.tryLater(ctx, delivery, fmt.Sprintf("a build result from %s", result.On), err) {
return return
} }
s.log.Printf("cannot keep a build result from %s: %v", result.On, err) s.log.Printf("cannot keep a build result from %s: %v", result.On, err)
s.settled(delivery) s.settled(subject, delivery)
_ = delivery.Reject(false) _ = delivery.Reject(false)
return return
} }
s.settled(delivery) s.settled(subject, delivery)
switch { switch {
case result.Failed != "": case result.Failed != "":
s.log.Printf("%s could not build %s", result.On, result.Repository) s.log.Printf("%s could not build %s", result.On, result.Repository)
@@ -518,13 +567,16 @@ func (s *Server) handleBuilt(ctx context.Context, delivery amqp.Delivery) {
// //
// Acknowledged after the work. A replay that fails for a reason other than the store is not one // Acknowledged after the work. A replay that fails for a reason other than the store is not one
// that succeeds by being handed the same request again, so that is acknowledged and said; but a // that succeeds by being handed the same request again, so that is acknowledged and said; but a
// store that could not be read right now is asked again after a pause, bounded, rather than the // store that could not be read right now is held and asked again, bounded, rather than the
// request lost until the catalogue next restarts (issue 083). // request lost until the catalogue next restarts (issue 083). One request stands for all: a newer
// one supersedes one still held.
func (s *Server) catchingUp(ctx context.Context, delivery amqp.Delivery) { func (s *Server) catchingUp(ctx context.Context, delivery amqp.Delivery) {
requeued := false const subject = "catch-up"
s.supersede(subject, delivery)
holding := false
defer func() { defer func() {
if !requeued { if !holding {
s.settled(delivery) s.settled(subject, delivery)
_ = delivery.Ack(false) _ = delivery.Ack(false)
} }
}() }()
@@ -534,8 +586,8 @@ func (s *Server) catchingUp(ctx context.Context, delivery amqp.Delivery) {
} }
announcements, err := s.replayer.Announceable(ctx) announcements, err := s.replayer.Announceable(ctx)
if err != nil { if err != nil {
if s.tryLater(ctx, delivery, "a catalogue's request to catch up", err) { if s.tryLater(delivery, subject, "a catalogue's request to catch up", err, s.catchingUp) {
requeued = true holding = true
return return
} }
s.log.Printf("a catalogue asked to catch up and the mesh could not read its builds: %v", err) s.log.Printf("a catalogue asked to catch up and the mesh could not read its builds: %v", err)
@@ -560,19 +612,24 @@ func (s *Server) catchingUp(ctx context.Context, delivery amqp.Delivery) {
// //
// **Acknowledged whatever happens, but one thing.** A failure here is usually the control plane // **Acknowledged whatever happens, but one thing.** A failure here is usually the control plane
// being unable to act on an upgrade — a machine that cannot be resolved, a broker that will not // being unable to act on an upgrade — a machine that cannot be resolved, a broker that will not
// take a declaration — and none of those get better by being handed the same message again. // take a declaration — and none of those get better by being handed the same message again. The
// Requeuing those would put a poison message at the head of a durable queue and stop every // one exception is the upgrader saying the store could not be read for the moment (ErrTryAgain):
// upgrade behind it. The one exception is the store unreachable for the moment, which does get // that is held and asked again, bounded (novox/hq issue 083). Only the upgrader's word counts
// better: that is asked again after a pause, for a bounded time (novox/hq issue 083). // here, not an error that merely looks like an outage — a push that timed out on the second
// machine is not asked again, or the first would be pushed every few seconds for two minutes.
// A newer move of the same module supersedes one still held.
func (s *Server) upgraded(ctx context.Context, delivery amqp.Delivery) { func (s *Server) upgraded(ctx context.Context, delivery amqp.Delivery) {
requeued := false var u Upgraded
_ = json.Unmarshal(delivery.Body, &u)
subject := "upgrade " + u.Module
s.supersede(subject, delivery)
holding := false
defer func() { defer func() {
if !requeued { if !holding {
s.settled(delivery) s.settled(subject, delivery)
_ = delivery.Ack(false) _ = delivery.Ack(false)
} }
}() }()
var u Upgraded
if err := json.Unmarshal(delivery.Body, &u); err != nil { if err := json.Unmarshal(delivery.Body, &u); err != nil {
s.log.Printf("an upgrade announcement could not be read: %v", err) s.log.Printf("an upgrade announcement could not be read: %v", err)
return return
@@ -582,11 +639,9 @@ func (s *Server) upgraded(ctx context.Context, delivery amqp.Delivery) {
return return
} }
if err := s.upgrader.Upgraded(ctx, u); err != nil { if err := s.upgrader.Upgraded(ctx, u); err != nil {
// A store that could not be read right now is asked again, bounded: acting on an upgrade if errors.Is(err, ErrTryAgain) &&
// twice pushes the same declarations twice, which converges (issue 083). Any other s.tryLater(delivery, subject, fmt.Sprintf("%s's move to %s", u.Module, short(u.Commit)), err, s.upgraded) {
// failure is acknowledged, as before — requeued, it would stop every upgrade behind it. holding = true
if s.tryLater(ctx, delivery, fmt.Sprintf("%s's move to %s", u.Module, short(u.Commit)), err) {
requeued = true
return return
} }
s.log.Printf("%s moved to %s and the mesh could not act on it: %v", s.log.Printf("%s moved to %s and the mesh could not act on it: %v",
+9 -5
View File
@@ -42,10 +42,13 @@ func a(t *testing.T, to *settledAs, key string, v any) amqp.Delivery {
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
return amqp.Delivery{Acknowledger: to, RoutingKey: key, Body: body} tag++
return amqp.Delivery{Acknowledger: to, RoutingKey: key, Body: body, DeliveryTag: tag}
} }
func quietServer() *Server { return &Server{log: log.New(io.Discard, "", 0), again: 1} } func quietServer() *Server { return &Server{log: log.New(io.Discard, "", 0)} }
func (a *settledAs) held() bool { return !a.acked && !a.nacked && !a.rejected }
// A build result the store could not take right now is handed back; one it refused is rejected, // A build result the store could not take right now is handed back; one it refused is rejected,
// as before; one it kept is acknowledged. // as before; one it kept is acknowledged.
@@ -56,7 +59,7 @@ func TestABuildResultWaitsOutARestartingStore(t *testing.T) {
err error err error
want func(*settledAs) bool want func(*settledAs) bool
}{ }{
{"restarting", restarting, func(s *settledAs) bool { return s.requeued && !s.acked && !s.rejected }}, {"restarting", restarting, func(s *settledAs) bool { return s.held() }},
{"refused", errors.New("no such module"), func(s *settledAs) bool { return s.rejected && !s.nacked }}, {"refused", errors.New("no such module"), func(s *settledAs) bool { return s.rejected && !s.nacked }},
{"kept", nil, func(s *settledAs) bool { return s.acked && !s.nacked }}, {"kept", nil, func(s *settledAs) bool { return s.acked && !s.nacked }},
} { } {
@@ -79,7 +82,8 @@ func TestAnUpgradeWaitsOutARestartingStoreAndNothingElse(t *testing.T) {
err error err error
want func(*settledAs) bool want func(*settledAs) bool
}{ }{
{"restarting", restarting, func(s *settledAs) bool { return s.requeued && !s.acked }}, {"the store away, said by the upgrader", errors.Join(ErrTryAgain, restarting), func(s *settledAs) bool { return s.held() }},
{"a push that timed out", context.DeadlineExceeded, func(s *settledAs) bool { return s.acked && !s.nacked }},
{"cannot act", errors.New("anchor cannot be resolved"), func(s *settledAs) bool { return s.acked && !s.nacked }}, {"cannot act", errors.New("anchor cannot be resolved"), func(s *settledAs) bool { return s.acked && !s.nacked }},
{"acted", nil, func(s *settledAs) bool { return s.acked && !s.nacked }}, {"acted", nil, func(s *settledAs) bool { return s.acked && !s.nacked }},
} { } {
@@ -101,7 +105,7 @@ func TestACatchUpWaitsOutARestartingStore(t *testing.T) {
err error err error
want func(*settledAs) bool want func(*settledAs) bool
}{ }{
{"restarting", restarting, func(s *settledAs) bool { return s.requeued && !s.acked }}, {"restarting", restarting, func(s *settledAs) bool { return s.held() }},
{"unreadable", errors.New("a build row is malformed"), func(s *settledAs) bool { return s.acked && !s.nacked }}, {"unreadable", errors.New("a build row is malformed"), func(s *settledAs) bool { return s.acked && !s.nacked }},
{"nothing to replay", nil, func(s *settledAs) bool { return s.acked && !s.nacked }}, {"nothing to replay", nil, func(s *settledAs) bool { return s.acked && !s.nacked }},
} { } {