Raise a check's store without durability and remove what earlier holders left (hq issue 306)
mesh/merge-gate pass: builds build-agent, mesh-controller, route-proxy → ace, g14, novox, shanks; no bus step; every machine composes with the change as it…
mesh/repo-check pass: its merge-check.sh passed
mesh/delivery delivered

On the control node the store-bound packages of the controller's suite ran
five to seven times slower than on any other holder, and every controller
check that landed there ran past the suite's thirty minutes: a throwaway
store flushing to a disk the mesh's own store, bus and forge keep busy.
A store that lives for one check needs no crash safety.

A holder recreated mid-check left the check's store and bus running, and
the redelivery went to another machine, so nothing removed them: eleven
pairs across four machines. A starting holder has taken nothing, so every
container labelled with an ask of the seat is an earlier holder's.
This commit is contained in:
jochen
2026-10-08 10:20:28 +02:00
parent c714d07739
commit 85b2a1855b
7 changed files with 178 additions and 1 deletions
+16
View File
@@ -274,6 +274,22 @@ func (h *holder) removeContainers(id string) (int, error) {
return builder.RemoveContainersOf(cleanup, h.remove, id)
}
// removeLeftBehind removes, once, what earlier holders on this machine left — called before anything is
// taken, so none of it is this holder's — and says what it did. A runtime that cannot be asked is said
// and the holder takes work all the same: what is left is waste, not a reason to stop building.
func (h *holder) removeLeftBehind() int {
cleanup, stop := context.WithTimeout(context.Background(), killRemoves)
defer stop()
n, err := builder.RemoveLeftBehind(cleanup, h.remove)
switch {
case err != nil:
fmt.Fprintf(os.Stderr, "could not look for containers earlier builds left on %s: %v\n", h.on, err)
case n > 0:
fmt.Fprintf(os.Stderr, "removed %d container(s) earlier builds left on %s\n", n, h.on)
}
return n
}
// returned records that the build's work ended, and says whether a kill came first — only then is
// the build killed; an error it ended with on its own is its own outcome.
func (h *holder) returned(r *running) bool {
+24
View File
@@ -3,6 +3,7 @@ package main
import (
"context"
"encoding/json"
"errors"
"strings"
"testing"
@@ -146,3 +147,26 @@ func TestAKillArrivingAfterTheBuildEndedIsRefusedAndAnUnannouncedKillSaysSo(t *t
t.Fatalf("kill said %q (%v)", said, err)
}
}
// **Issue 306**: a holder starting removes the containers earlier builds left on its machine, through
// the runner a kill uses, and a runtime that cannot be asked does not stop it.
func TestAHolderStartingRemovesWhatEarlierBuildsLeftHere(t *testing.T) {
h := newHolder("novox", link.TheBuildMachine, t.TempDir(), nil)
var removed []string
h.remove = func(_ context.Context, _ string, name string, args ...string) (string, error) {
removed = append(removed, name+" "+strings.Join(args, " "))
if args[0] == "ps" {
return "c1 build-1\nc2 build-1\nc3 check-here-2\n", nil
}
return "", nil
}
if n := h.removeLeftBehind(); n != 2 || len(removed) != 2 || removed[1] != "docker rm -f c1 c2" {
t.Fatalf("removed %d: %v", n, removed)
}
h.remove = func(context.Context, string, string, ...string) (string, error) {
return "", errors.New("no runtime")
}
if n := h.removeLeftBehind(); n != 0 {
t.Errorf("a runtime that cannot be asked removed %d", n)
}
}
+3
View File
@@ -129,6 +129,9 @@ func run() error {
if h.Paused() {
fmt.Fprintf(os.Stderr, "paused (kept in %s): taking no build until resumed\n", pausedFile(workspace))
}
// **What an earlier holder here left goes before anything is taken** (novox/hq issue 306): a holder
// recreated mid-check left the check's store and bus running, and the ask went to another machine.
h.removeLeftBehind()
machine := link.MachineOverNATSWith(js, on, seat, link.MachineOptions{Paused: h.Paused})
defer machine.Close()