Let two controllers overlap safely while one hands over to the other (hq issue 213)

The controller's machine moves it from the container to a process by
starting the process first and removing the container once the process
is up (mesh-host's `replaces`). For that moment two controllers share the
store and the bus. Checked what each does:

- the seat's verbs: a queue group per seat, each call answered once. Safe.
- the controller's consumers on CONTROL and EVENTS: push consumers with
  no delivery group, so the second bind is refused with "consumer is
  already bound" and serve exited. The process would restart for ever,
  the host would never see it up, and the container would never go. The
  second controller now stands by and binds when the first lets go
  (tested on a real bus; fails without the change).
- plans: read, changed and saved whole by the 30s timer, by build
  outcomes, by a merge and by `plans stop`. Two timers would each ask a
  tier the other had just asked. Working the plans now takes a
  session-level advisory lock on the inventory: the timer skips while
  another holds it, the other paths wait for it. Build asks happen only
  inside plan work and are covered by the same lock.
This commit is contained in:
jochen
2026-10-04 01:11:26 +02:00
parent 7bb9e55d0b
commit e11caecdad
7 changed files with 315 additions and 3 deletions
+41 -1
View File
@@ -2,6 +2,7 @@ package main
import (
"context"
"errors"
"flag"
"fmt"
"sort"
@@ -296,6 +297,15 @@ func askTier(ctx context.Context, inv *inventory.Inventory, p *inventory.Plan) e
// build's request time is not known, and such an outcome is taken as before.
func planBuilt(ctx context.Context, open *stores, module, commit, failed string, asked time.Time) {
inv := open.inventory
// One controller works the plans at a time (novox/hq issue 213); an outcome waits its turn rather
// than write over what the holder is about to save. Not taken, it is still in the build records,
// which the holder settles the plan from (issue 214).
release, err := inv.HoldPlans(ctx, true)
if err != nil {
fmt.Printf("plans: %s's outcome is left to the build records: %v\n", module, err)
return
}
defer release()
plans, err := inv.OpenPlans(ctx)
if err != nil {
fmt.Printf("plans: cannot read them: %v\n", err)
@@ -342,13 +352,30 @@ func planBuilt(ctx context.Context, open *stores, module, commit, failed string,
fmt.Printf("%s: %s; the tiers after it are not asked\n", p.ID, p.Note)
}
}
advancePlans(ctx, open)
advanceHeld(ctx, open)
}
// advancePlans moves every open plan as far as the facts allow: a tier whose modules are all built
// and whose gates are applied gives way to the next; the last tier done is the plan done. Called
// after every outcome and on a timer, so a plan waiting on a machine's report moves when it comes.
//
// **One controller at a time** (novox/hq issue 213). A plan is read, changed and saved whole; two
// controllers — the old and the new while a machine hands its controller over — would each ask a
// tier the other had just asked. Taken without waiting: whoever holds the plans is moving them.
func advancePlans(ctx context.Context, open *stores) {
release, err := open.inventory.HoldPlans(ctx, false)
if err != nil {
if !errors.Is(err, inventory.ErrPlansBusy) {
fmt.Printf("plans: cannot hold them: %v\n", err)
}
return
}
defer release()
advanceHeld(ctx, open)
}
// advanceHeld is advancePlans for a caller already holding the plans.
func advanceHeld(ctx context.Context, open *stores) {
inv := open.inventory
plans, err := inv.OpenPlans(ctx)
if err != nil {
@@ -676,6 +703,19 @@ func plansCommand(ctx context.Context, args []string) error {
}
p.State = inventory.PlanFailed
p.Note = "stopped by hand at tier " + fmt.Sprint(p.Tier)
release, err := inv.HoldPlans(ctx, true)
if err != nil {
return err
}
defer release()
if p, err = inv.PlanByID(ctx, positionals[1]); err != nil {
return err
}
if !p.Open() {
return fmt.Errorf("%s is already %s", p.ID, p.State)
}
p.State = inventory.PlanFailed
p.Note = "stopped by hand at tier " + fmt.Sprint(p.Tier)
if err := inv.SavePlan(ctx, p); err != nil {
return err
}