Merge main: hold a moving machine back, then deliver the rest grants first

Main sends every machine through deliver (memberships before declarations,
hq ADR 0218); this branch holds back a machine whose running module's data
would move (ADR 0217). Both stand: the held machines are filtered out before
delivery, in the named push and its cascade alike.
This commit is contained in:
jochen
2026-10-05 18:47:52 +02:00
37 changed files with 2821 additions and 205 deletions
+53 -1
View File
@@ -31,7 +31,12 @@ func collect(ctx context.Context, inv *inventory.Inventory) {
fmt.Fprintf(os.Stderr, "could not work out what the artifact store may let go of: %v\n", err)
return
}
if len(references) == 0 {
kept, err := inv.KeptArchives(ctx)
if err != nil {
fmt.Fprintf(os.Stderr, "could not work out which archives the artifact store keeps: %v\n", err)
return
}
if len(references) == 0 && len(kept) == 0 {
return
}
shelf, err := inv.Catalogue(ctx)
@@ -59,6 +64,25 @@ func collect(ctx context.Context, inv *inventory.Inventory) {
defer stop()
store := artifacts.Store{Address: address}
// **Hold before letting go** (novox/hq issue 253). The store's collector keeps only what a
// manifest names, and archives were published as bare blobs, so every kept archive is first
// held by its manifest — which backfills the ones published before holders, a few at a time
// as builds come, and is two HEADs each once done. A kept archive that could not be held stops
// the sweep before it deletes anything: "everything kept is held" is the precondition the
// collector's safety rests on, and a store refusing a hold would refuse the deletes too.
wrote, missing, err := holdKept(within, store, kept)
if wrote > 0 {
fmt.Fprintf(os.Stderr, "the artifact store now holds %d more kept archive(s) by a manifest\n", wrote)
}
if missing > 0 {
fmt.Fprintf(os.Stderr, "%d archive(s) the mesh keeps are not in the artifact store at all; "+
"`collection` lists them\n", missing)
}
if err != nil {
fmt.Fprintf(os.Stderr, "not every kept archive could be held, so nothing was let go: %v\n", err)
return
}
var done []string
var left, skipped int
for i, reference := range references {
@@ -114,6 +138,34 @@ func collect(ctx context.Context, inv *inventory.Inventory) {
}
}
// holdKept holds every kept archive by its manifest, stopping at the first refusal by the store.
// Answers how many holders it wrote and how many kept archives the store does not have.
//
// A missing archive is counted rather than fatal: there is nothing to hold, and that is a fact
// for an operator to read (`collection`), not a reason to stop collecting what is not kept. A
// reference the store cannot be asked about is skipped as the deletion loop skips one
// (novox/hq issue 226).
func holdKept(ctx context.Context, store artifacts.Store, kept []string) (wrote, missing int, err error) {
for _, reference := range kept {
if err := ctx.Err(); err != nil {
return wrote, missing, fmt.Errorf("ran out of time before %s: %w", reference, err)
}
did, err := store.Hold(ctx, reference)
switch {
case err == nil:
if did {
wrote++
}
case errors.Is(err, artifacts.Gone):
missing++
case errors.Is(err, artifacts.ErrNotOurs):
default:
return wrote, missing, fmt.Errorf("holding %s: %w", reference, err)
}
}
return wrote, missing, nil
}
// mostPerSweep is how many artifacts one sweep will ask about. Enough that a mesh building
// several times a day converges within days of this landing; small enough that no single build
// waits on the whole backlog.
+166
View File
@@ -0,0 +1,166 @@
package main
import (
"context"
"encoding/json"
"errors"
"flag"
"fmt"
"os"
"strings"
"github.com/novox/mesh-controller/internal/artifacts"
)
// What the artifact store keeps, whether each kept archive is held, and what the sweep may let go
// (novox/hq issue 253, ADR 0189).
//
// **The question to answer before the store's collector runs for real.** The collector deletes
// every blob no manifest names, and archives were published as bare blobs, so the nightly step
// runs `--dry-run` until every archive the mesh keeps is held by its manifest. The sweep holds
// them as builds come; this says how far that has got — "0 unheld" is the number that lets the
// dry run go.
//
// Reads and changes nothing: each kept archive is asked about with HEADs only. Reached over the
// console through the mesh-controller seat's `command` verb (`collection --json`), which needs no
// new verb in the seat's row.
type collectionReport struct {
// Store is the artifact store as this machine reached it; empty when it is not on the network.
Store string `json:"store"`
// KeptArchives is how many archives the mesh keeps, for either reason.
KeptArchives int `json:"kept_archives"`
// Held is how many of them the store holds by their manifest.
Held int `json:"held"`
// Unheld are the kept archives the collector would delete tonight if it ran for real.
Unheld []string `json:"unheld"`
// Missing are kept archives the store does not have at all.
Missing []string `json:"missing"`
// Unasked is how many could not be asked about, and why the asking stopped.
Unasked int `json:"unasked"`
Stopped string `json:"stopped,omitempty"`
// Eligible is what the sweep may let go of: made by the mesh, kept for no reason, not yet
// collected — split by kind.
Eligible int `json:"eligible"`
EligibleImages int `json:"eligible_images"`
EligibleArchives int `json:"eligible_archives"`
// SafeToCollect is whether every kept archive was asked about and every one is held.
SafeToCollect bool `json:"safe_to_collect"`
}
func collectionCommand(ctx context.Context, args []string) error {
set := flag.NewFlagSet("collection", flag.ContinueOnError)
asJSON := set.Bool("json", false, "answer as JSON")
positionals, err := parseAround(set, args)
if err != nil {
return err
}
if len(positionals) != 0 {
return errors.New("collection [--json]")
}
open, err := openStores(ctx)
if err != nil {
return err
}
defer open.Close()
inv := open.inventory
kept, err := inv.KeptArchives(ctx)
if err != nil {
return err
}
eligible, err := inv.ToCollect(ctx)
if err != nil {
return err
}
report := collectionReport{KeptArchives: len(kept), Eligible: len(eligible), Unheld: []string{}, Missing: []string{}}
for _, reference := range eligible {
if strings.Contains(reference, "/blobs/") {
report.EligibleArchives++
} else {
report.EligibleImages++
}
}
shelf, err := inv.Catalogue(ctx)
if err != nil {
return err
}
report.Store, err = artifactStoreAddress(ctx, inv, shelf, "")
if err != nil {
return err
}
if report.Store == "" {
report.Unasked = len(kept)
report.Stopped = "this mesh has no artifact store on its network"
} else {
report.Unasked, report.Stopped = askHeld(ctx, artifacts.Store{Address: report.Store}, kept, &report)
}
report.SafeToCollect = report.Unasked == 0 && len(report.Unheld) == 0
if *asJSON {
encoder := json.NewEncoder(os.Stdout)
encoder.SetIndent("", " ")
return encoder.Encode(report)
}
printCollection(report)
return nil
}
// askHeld asks the store about each kept archive, stopping at the first answer that is not about
// the archive: a store that cannot be reached for one cannot be for the next, and a page of
// identical failures says less than one line.
func askHeld(ctx context.Context, store artifacts.Store, kept []string, report *collectionReport) (int, string) {
for i, reference := range kept {
held, err := store.Held(ctx, reference)
switch {
case err == nil && held:
report.Held++
case err == nil:
report.Unheld = append(report.Unheld, reference)
case errors.Is(err, artifacts.Gone):
report.Missing = append(report.Missing, reference)
case errors.Is(err, artifacts.ErrNotOurs):
// KeptArchives names only the mesh's own; counted as unasked if one ever is not.
report.Unasked++
default:
return report.Unasked + len(kept) - i, fmt.Sprintf("asking about %s: %v", reference, err)
}
}
return report.Unasked, ""
}
func printCollection(r collectionReport) {
store := r.Store
if store == "" {
store = "(not on the network)"
}
fmt.Printf("artifact store %s\n", store)
fmt.Printf("kept archives %d\n", r.KeptArchives)
fmt.Printf(" held %d\n", r.Held)
fmt.Printf(" unheld %d\n", len(r.Unheld))
fmt.Printf(" missing %d\n", len(r.Missing))
if r.Unasked > 0 {
fmt.Printf(" not asked %d (%s)\n", r.Unasked, r.Stopped)
}
fmt.Printf("eligible to let go %d (%d images, %d archives)\n", r.Eligible, r.EligibleImages, r.EligibleArchives)
if len(r.Unheld) > 0 {
fmt.Println("\nunheld — the store's collector would delete these; the next build's sweep holds them:")
for _, reference := range r.Unheld {
fmt.Printf(" %s\n", reference)
}
}
if len(r.Missing) > 0 {
fmt.Println("\nmissing — kept by the mesh, not in the store:")
for _, reference := range r.Missing {
fmt.Printf(" %s\n", reference)
}
}
fmt.Println()
if r.SafeToCollect {
fmt.Println("every kept archive is held: the store's collector may run for real")
} else {
fmt.Println("NOT every kept archive is known to be held: keep the store's collector on --dry-run")
}
}
+145
View File
@@ -0,0 +1,145 @@
package main
import (
"context"
"errors"
"reflect"
"strings"
"testing"
"github.com/novox/mesh-controller/internal/link"
)
// recordingDelivery is a delivery that writes down what was done, in order, and fails where told.
type recordingDelivery struct {
did []string
grantErr error
declareErr error
}
func (r *recordingDelivery) grant(_ context.Context, sending []readyNode) error {
for _, s := range sending {
r.did = append(r.did, "grant "+s.node)
}
return r.grantErr
}
func (r *recordingDelivery) declare(_ context.Context, s readyNode, _ []byte) (string, error) {
if r.declareErr != nil {
return "", r.declareErr
}
r.did = append(r.did, "declare "+s.node)
return "digest-" + s.node, nil
}
func ready(names ...string) []readyNode {
var out []readyNode
for _, n := range names {
out = append(out, readyNode{node: n, declared: sendable{Resources: []map[string]any{{"id": "x"}}}})
}
return out
}
// novox/hq issue 249: the grants that come with a module's new declarations are issued before any
// machine is sent the code that uses them — except the machine holding the bus, whose declaration
// carries the controller's own right to issue them, and goes first.
func TestGrantsAreIssuedBeforeTheDeclarations(t *testing.T) {
d := &recordingDelivery{}
digests, err := deliver(t.Context(), d, "", ready("anchor", "laptop"))
if err != nil {
t.Fatal(err)
}
want := []string{"grant anchor", "grant laptop", "declare anchor", "declare laptop"}
if !reflect.DeepEqual(d.did, want) {
t.Fatalf("delivered in the order %v, wanted %v", d.did, want)
}
if digests["anchor"] != "digest-anchor" || digests["laptop"] != "digest-laptop" {
t.Fatalf("the digests sent were not answered: %v", digests)
}
held := &recordingDelivery{}
if _, err := deliver(t.Context(), held, "broker", ready("broker", "anchor")); err != nil {
t.Fatal(err)
}
want = []string{"declare broker", "grant broker", "grant anchor", "declare anchor"}
if !reflect.DeepEqual(held.did, want) {
t.Fatalf("with the bus's machine in the send: %v, wanted %v", held.did, want)
}
}
// A grant that cannot be issued holds back the machines it concerns and is an error the caller
// retries on — never "until the next push" — and the bus's own machine is sent regardless, so the
// grant that would let the controller issue memberships is never held behind them.
func TestAGrantThatFailsHoldsBackWhatItConcerns(t *testing.T) {
// A failure naming no machine (the buckets): everything but the bus's machine.
d := &recordingDelivery{grantErr: errors.New("the bus refused the bucket")}
_, err := deliver(t.Context(), d, "broker", ready("broker", "anchor", "laptop"))
if err == nil || !errors.Is(err, errGrants) || !strings.Contains(err.Error(), "the bus refused the bucket") ||
!strings.Contains(err.Error(), "anchor, laptop not sent") {
t.Fatalf("a failed grant was not said as the send's failure: %v", err)
}
if want := []string{"declare broker", "grant broker", "grant anchor", "grant laptop"}; !reflect.DeepEqual(d.did, want) {
t.Fatalf("delivered %v, wanted the bus's machine alone", d.did)
}
// A membership that failed for one machine: that machine alone.
one := &recordingDelivery{grantErr: &grantsRefused{nodes: map[string]error{"laptop": errors.New("no")}}}
_, err = deliver(t.Context(), one, "", ready("anchor", "laptop"))
if !errors.Is(err, errGrants) || !strings.Contains(err.Error(), "laptop not sent") {
t.Fatalf("one machine's refused membership was not said: %v", err)
}
if want := []string{"grant anchor", "grant laptop", "declare anchor"}; !reflect.DeepEqual(one.did, want) {
t.Fatalf("delivered %v, wanted anchor sent and laptop held back", one.did)
}
// The announced upgrade that hit it is asked again.
if !errors.Is(askAgainOnGrants(err), link.ErrTryAgain) {
t.Fatal("an announcement whose send stopped at its grants is not asked again")
}
if other := errors.New("laptop could not be resolved"); errors.Is(askAgainOnGrants(other), link.ErrTryAgain) {
t.Fatal("any failure is asked again, not only a grant's")
}
// Nothing to send is nothing granted either.
none := &recordingDelivery{grantErr: errors.New("never asked")}
if _, err := deliver(t.Context(), none, "", nil); err != nil || len(none.did) != 0 {
t.Fatalf("an empty send granted or failed: %v %v", none.did, err)
}
}
// Whether the bus's machine goes first is read from the user list alone, by its digest.
func TestTheBusMachineIsBehindByItsUserListAlone(t *testing.T) {
list := "users: [a, b]"
if userListBehind(list, digestOf([]byte(list))) {
t.Fatal("the list it was sent reads as behind")
}
if !userListBehind(list, digestOf([]byte("users: [a]"))) || !userListBehind(list, "") {
t.Fatal("a changed or never-sent list reads as current")
}
if userListBehind("", "") {
t.Fatal("a machine sent no list reads as behind")
}
}
// The machine holding the bus goes first: its declaration carries the user list the new grants are
// checked against. Among the machines it is moved to the front; not among them it is added only
// when it is behind.
func TestTheMachineHoldingTheBusIsSentFirst(t *testing.T) {
for _, c := range []struct {
what string
names []string
holder string
behind bool
want []string
}{
{"among them", []string{"ace", "g14", "novox"}, "novox", false, []string{"novox", "ace", "g14"}},
{"not among them, behind", []string{"ace", "g14"}, "novox", true, []string{"novox", "ace", "g14"}},
{"not among them, current", []string{"ace", "g14"}, "novox", false, []string{"ace", "g14"}},
{"nothing holds the bus", []string{"ace", "g14"}, "", true, []string{"ace", "g14"}},
{"only it", []string{"novox"}, "novox", false, []string{"novox"}},
} {
if got := brokerFirst(c.names, c.holder, c.behind); !reflect.DeepEqual(got, c.want) {
t.Errorf("%s: sent in the order %v, wanted %v", c.what, got, c.want)
}
}
}
+4 -4
View File
@@ -18,16 +18,16 @@ func TestASendRoundGivesItsHoldBackOnEveryWayOut(t *testing.T) {
plain := func(context.Context, string) (sendable, error) {
return sendable{Resources: []map[string]any{{"id": "x"}}}, nil
}
failing := func(readyNode, []byte) error { return errors.New("the broker went away") }
fine := func(readyNode, []byte) error { return nil }
failing := &recordingDelivery{declareErr: errors.New("the broker went away")}
fine := &recordingDelivery{}
for name, round := range map[string]func() error{
"a body that cannot be marshalled": func() error {
_, err := sendRound(ctx, open, []string{"anchor"}, unmarshallable, fine)
_, err := sendRound(ctx, open, []string{"anchor"}, unmarshallable, fine, "", nil)
return err
},
"a send that fails": func() error {
_, err := sendRound(ctx, open, []string{"anchor"}, plain, failing)
_, err := sendRound(ctx, open, []string{"anchor"}, plain, failing, "", nil)
return err
},
} {
+3
View File
@@ -73,6 +73,8 @@ func run() error {
return askCommand(ctx, args[1:])
case "builds":
return buildsCommand(ctx, args[1:])
case "collection":
return collectionCommand(ctx, args[1:])
case "plans":
return plansCommand(ctx, args[1:])
case "pin":
@@ -200,6 +202,7 @@ func usage() {
build --behind build every module the mesh holds older than its source
build --on <module> rebuild every module that stands on this module's artifacts, bases first
builds [<module>] what has been built lately, and what came of it
collection [--json] kept archives held/unheld by a manifest, and what the sweep may let go
builder issue <name> a broker account for a build machine, scoped to build work,
delivered as the builder module's broker secret (module add it first)
licence add|list|use|key model access, under the name a person calls it
+33 -1
View File
@@ -386,8 +386,12 @@ func brokerCommand(ctx context.Context, args []string) error {
if len(args) > 0 && args[0] == "accounts" {
return busAccounts(ctx, args[1:])
}
if len(args) > 0 && args[0] == "consumer-reset" {
return consumerReset(args[1:])
}
if len(args) == 0 || args[0] != "show" {
return errors.New("broker show | broker certificate [--check] --into <directory> | broker accounts --into <file>")
return errors.New("broker show | broker certificate [--check] --into <directory> | broker accounts --into <file> | " +
"broker consumer-reset <stream> <consumer>")
}
known, err := broker.FromEnvironment()
if errors.Is(err, broker.ErrNotConfigured) {
@@ -407,6 +411,34 @@ func brokerCommand(ctx context.Context, args []string) error {
return nil
}
// consumerReset re-makes one consumer on a stream that keeps history to start from now (novox/hq issue
// 248): the way out of a consumer replaying a week of announcements, said rather than done by hand. A
// person's act — what was pending is dropped — so it is a command, and nothing calls it on its own.
func consumerReset(args []string) error {
if len(args) != 2 {
return errors.New("broker consumer-reset <stream> <consumer>, e.g. broker consumer-reset EVENTS controller")
}
address, err := broker.BusAddress()
if err != nil {
return err
}
js, err := broker.Dial(address)
if err != nil {
return fmt.Errorf("cannot reach the bus: %w", err)
}
defer js.Close()
before, after, err := js.ResetConsumer(args[0], args[1])
if err != nil {
return err
}
fmt.Printf("consumer %s on %s re-made to deliver from now\n", args[1], args[0])
fmt.Printf(" before: delivers %s, delivered to %d, acknowledged to %d, %d pending, %d unacknowledged\n",
before.DeliverPolicy, before.Delivered, before.AckFloor, before.Pending, before.AckPending)
fmt.Printf(" after: delivers %s, %d pending; what was pending is dropped. A holder bound to it may need its "+
"process restarted to bind again\n", after.DeliverPolicy, after.Pending)
return nil
}
// heardFrom says when a node was last heard from, in a form somebody can act on.
//
// "never" and "an hour ago" are different answers and are kept different. A node that has never
+8
View File
@@ -147,6 +147,14 @@ func TestAMergeRebuildsTheModulesItChanged(t *testing.T) {
{"a module the mesh does not hold", merge([]string{"modules/plex/index.ts"}, false), ""},
{"nothing said about the files", merge(nil, false), "gitea,keycloak"},
{"more files than were listed", merge([]string{"modules/gitea/index.ts"}, true), "gitea,keycloak"},
// novox/hq issue 252: a module the mesh has never registered is still a module, when the merge
// shows it is one — and a directory that may be shared code is still shared.
{"a new module beside a held one", merge([]string{"modules/gitea/x", "modules/newmod/module.json"}, false), "gitea"},
{"a new module's other files", merge([]string{"modules/newmod/index.ts", "modules/newmod/module.json"}, false), ""},
{"a module removed", merge([]string{"modules/gone/module.json"}, false), ""},
{"a directory with no manifest", merge([]string{"modules/lib/x.go"}, false), "gitea,keycloak"},
{"a file directly among the modules", merge([]string{"modules/README.md"}, false), "gitea,keycloak"},
{"the root's files still", merge([]string{"tsconfig.json"}, false), "gitea,keycloak"},
} {
if got := named(whatTheMergeTouched(candidates, known, c.m)); got != c.want {
t.Errorf("%s: rebuilt %q, wanted %q", c.what, got, c.want)
+22 -12
View File
@@ -398,7 +398,7 @@ func declarationWith(ctx context.Context, open *stores, node string,
return sendable{}, err
}
return sendable{Resources: composed.Resources, Adoption: adoption,
Received: composed.Received, Mesh: with.Mesh,
Received: composed.Received, Mesh: with.Mesh, BusUsers: with.BusUsers,
LeftOut: sortedKeysOf(composed.LeftOut), leftOutWhy: composed.LeftOut}, nil
}
@@ -1355,6 +1355,20 @@ func composeBusUsers(ctx context.Context, inv *inventory.Inventory,
//
// Asked of what this push resolves to rather than of the seat's holder mesh-wide: the file is a
// resource of that module, so the question is whether it is here.
list, missing, err := busUserList(ctx, inv, onThisNode)
if len(missing) > 0 {
fmt.Printf("the bus's user list leaves out %d user(s) the mesh has minted no credential "+
"for: %s. Each is a user that cannot connect until one is issued\n",
len(missing), strings.Join(missing, ", "))
}
return list, err
}
// busUserList is composeBusUsers without saying anything: the list, and the users left out of it
// for want of a credential. Asked on every send to decide whether the machine holding the bus must
// go first (novox/hq issue 249), where saying the same missing users each time would bury them.
func busUserList(ctx context.Context, inv *inventory.Inventory,
onThisNode []catalogue.Manifest) (string, []string, error) {
holdsTheBus := false
for _, m := range onThisNode {
if m.BusUsers != "" && m.ClaimsSeat("mesh-broker") {
@@ -1362,37 +1376,33 @@ func composeBusUsers(ctx context.Context, inv *inventory.Inventory,
}
}
if !holdsTheBus {
return "", nil
return "", nil, nil
}
records, err := inv.BusRecords(ctx)
if err != nil {
return "", err
return "", nil, err
}
users, err := broker.Users(records)
if err != nil {
return "", err
return "", nil, err
}
kept, err := inv.BusUsers(ctx)
if err != nil {
return "", err
return "", nil, err
}
hashes := make(map[string]string, len(kept))
for name, u := range kept {
hashes[name] = u.PasswordHash
}
filled, missing := broker.WithPasswords(users, hashes)
if len(missing) > 0 {
fmt.Printf("the bus's user list leaves out %d user(s) the mesh has minted no credential "+
"for: %s. Each is a user that cannot connect until one is issued\n",
len(missing), strings.Join(missing, ", "))
}
if len(filled) == 0 {
return "", fmt.Errorf(
return "", missing, fmt.Errorf(
"this machine runs the bus and not one user has a credential, so the composed list " +
"would refuse every connection in the mesh")
}
return broker.ComposeAccounts(filled)
list, err := broker.ComposeAccounts(filled)
return list, missing, err
}
// providerModuleOf is which module answers a need on the providing node: the one in this node's
+341 -94
View File
@@ -375,6 +375,17 @@ func pushCommand(ctx context.Context, args []string) error {
asked = append(asked, n.Name)
}
// **The machine holding the bus first** (novox/hq issue 249): its declaration carries the bus's
// user list, and a module's new grants are refused by the bus until that list says them. Among
// the machines asked it goes first; not among them and behind, it is added — a named push whose
// module gained a state would otherwise send the code and leave the right to use it for the
// cascade below, after.
holder, holderBehind, err := brokerBehind(ctx, open, asked)
if err != nil {
return err
}
asked = brokerFirst(asked, holder, holderBehind)
// Held from composing to sending, so a converge on one of them cannot send between the two
// and be overtaken by what was composed before it (novox/hq ADR 0100).
held, release, err := holdNodes(ctx, open, asked)
@@ -404,43 +415,22 @@ func pushCommand(ctx context.Context, args []string) error {
return declared, err
})
sentDigest := map[string]string{}
defer release()
for _, s := range sending {
// The number is inside the signed bytes, so a replayed older declaration cannot borrow a
// newer one's (novox/hq 04-ISSUES/107); it was taken when the composition began (issue 204).
body, err := s.declared.Body()
if err != nil {
return err
}
// A running module's data moving holds this machine, and only this one (novox/hq ADR 0217).
if moves, err := heldMoves(ctx, inv, s.node, body, movable); err != nil {
return err
} else if len(moves) > 0 {
sayHeld(os.Stdout, s.node, moves)
heldBack = append(heldBack, s.node)
continue
}
if err := link.Declare(ctx, server.Bus(), ident, s.node, body, 15*time.Second); err != nil {
return err
}
// After it is away, not before. A digest recorded for something that failed to send would
// make the machine look current for a declaration it never received.
digest, err := recordSent(ctx, inv, s.node, body)
if err != nil {
return err
}
sentDigest[s.node] = digest
fmt.Printf("sent %s %d resource(s)\n", s.node, len(s.declared.Resources))
// A machine whose declaration would move a running module's data is held, and only it (novox/hq
// ADR 0217); every other machine is delivered, its memberships first, then its declaration
// (novox/hq issue 249, ADR 0218).
bus := overTheBus{open: open, server: server, signer: ident}
toSend, err := holdingBack(ctx, inv, sending, movable, &heldBack)
if err != nil {
return err
}
sentDigest, err := deliver(ctx, bus, holder, toSend)
if err != nil {
return err
}
release()
fmt.Printf("\n%d node(s) told\n", len(sending))
reportUnheldPushed(os.Stdout, len(args) == 1, asked, unheld)
// And each machine's memberships, as every other send does (ADR 0160): a push is the one most
// operators run, and on 2026-10-01 it was the one path that issued none.
if err := issueMemberships(ctx, open, server, sending); err != nil {
return err
}
// **A named push leaves the mesh consistent, not just the machine it named** (novox/hq
// issue 057, ADR 0083). Assigning a cross-node consumer mints a provision, and the PROVIDER's
@@ -509,24 +499,9 @@ func pushCommand(ctx context.Context, args []string) error {
}
return declared, err
},
func(s readyNode, body []byte) error {
bus, holder, func(held context.Context, sending []readyNode) ([]readyNode, error) {
// The cascade is a push too, and held the same way (novox/hq ADR 0217).
if moves, err := heldMoves(ctx, inv, s.node, body, movable); err != nil {
return err
} else if len(moves) > 0 {
sayHeld(os.Stdout, s.node, moves)
heldBack = append(heldBack, s.node)
return nil
}
if err := link.Declare(ctx, server.Bus(), ident, s.node, body,
15*time.Second); err != nil {
return err
}
if _, err := recordSent(ctx, inv, s.node, body); err != nil {
return err
}
fmt.Printf("sent %s %d resource(s)\n", s.node, len(s.declared.Resources))
return nil
return holdingBack(held, inv, sending, movable, &heldBack)
})
refusals = append(refusals, refused...)
if err != nil {
@@ -659,10 +634,10 @@ func composeEach(names []string, allot func(node string) (int64, error),
// sendRound holds the named nodes, composes each and sends each that composed, and gives the hold
// back on every way out — a body that cannot be marshalled and a send that fails included
// (novox/hq ADR 0100). A node that cannot be composed is a refusal, not an error: the others are
// still sent.
// still sent. Their memberships go before their declarations, as every send's do (issue 249).
func sendRound(ctx context.Context, open *stores, names []string,
compose func(held context.Context, node string) (sendable, error),
send func(s readyNode, body []byte) error) ([]string, error) {
d delivery, holder string, keep func(context.Context, []readyNode) ([]readyNode, error)) ([]string, error) {
held, release, err := holdNodes(ctx, open, names)
if err != nil {
return nil, err
@@ -671,18 +646,276 @@ func sendRound(ctx context.Context, open *stores, names []string,
sending, refused := composeEach(names, allotting(held, open.inventory), func(node string) (sendable, error) {
return compose(held, node)
})
for _, s := range sending {
body, err := s.declared.Body()
if err != nil {
return refused, err
}
if err := send(s, body); err != nil {
if keep != nil {
if sending, err = keep(held, sending); err != nil {
return refused, err
}
}
if _, err := deliver(held, d, holder, sending); err != nil {
return refused, err
}
return refused, nil
}
// holdingBack leaves out each machine whose declaration would move a running module's data, says so,
// and names it among those held back (novox/hq ADR 0217); the rest go on to be delivered, grants first
// (ADR 0218). A declaration's body is the same bytes however often it is read.
func holdingBack(ctx context.Context, inv *inventory.Inventory, sending []readyNode, movable map[string]bool,
heldBack *[]string) ([]readyNode, error) {
var out []readyNode
for _, s := range sending {
body, err := s.declared.Body()
if err != nil {
return nil, err
}
moves, err := heldMoves(ctx, inv, s.node, body, movable)
if err != nil {
return nil, err
}
if len(moves) > 0 {
sayHeld(os.Stdout, s.node, moves)
*heldBack = append(*heldBack, s.node)
continue
}
out = append(out, s)
}
return out, nil
}
// delivery is the two acts of sending machines what they should be, apart, so the order between
// them is one function's and can be read and tested there (novox/hq issue 249).
type delivery interface {
// grant issues what the machines' modules may do — each module's state raised and its
// membership issued — for every machine about to be sent.
grant(ctx context.Context, sending []readyNode) error
// declare sends one machine its declaration and records it sent, answering the digest.
declare(ctx context.Context, s readyNode, body []byte) (string, error)
}
// errGrants marks a send that stopped because what the machines' modules may do could not be issued
// (novox/hq issue 249). Nothing about the machines is wrong; asked again, it is likely to work, so an
// announcement that hits it is held and asked again.
var errGrants = errors.New("what the machines' modules may do on the bus could not be issued, and code " +
"sent before its grants is refused there")
// grantsRefused is a grant that failed for some machines and not others: their memberships could not
// be issued, by machine, and only those machines are held back.
type grantsRefused struct{ nodes map[string]error }
func (g *grantsRefused) Error() string {
names := make([]string, 0, len(g.nodes))
for n := range g.nodes {
names = append(names, n)
}
sort.Strings(names)
return fmt.Sprintf("the memberships of %s could not be issued; the first: %v",
strings.Join(names, ", "), g.nodes[names[0]])
}
// deliver sends the machines their declarations: **the machine holding the bus, then the grants,
// then the rest** (novox/hq issue 249).
//
// A merge gave a module a new state; its bundle reached every machine within a minute, and the
// machines' permissions on the bus did not include the state until somebody pushed by hand: the code
// arrived before the right to use it. A module that read its new state on start failed its start; the
// one that was there retried for two minutes. The memberships were issued after the declarations —
// "because the runtime it is for arrives with it" — and a membership is retained last-per-subject on
// the bus (internal/link/bus.go), so issued first it waits for the runtime that arrives after it. A
// runtime still on the old code merely holds a grant it does not use yet.
//
// **The holder's declaration before the grants, though.** The bus's user list travels in it, and the
// controller's own right to publish memberships and raise buckets is in that list (the precedent of
// issue 183): grants first, and a grant the controller is not yet allowed to make would hold the very
// declaration that allows it — a lock only a hand on the broker could open. A runtime already running
// on that machine follows a membership issued after its declaration, as it always has.
//
// **A grant that cannot be issued holds back what it concerns, and says so as an error.** It used to
// be said and passed over — "the machines keep what they derive until the next push" — which reported
// a rollout done that had delivered code its machines could not run. A membership that failed holds
// back its own machine; a failure that names no machine (the buckets) holds back every machine but the
// holder, already sent. The error carries errGrants, so the caller's rollout is not marked sent and is
// tried again.
//
// The grants are issued, not waited on: a membership is a retained message the runtime reads when it
// comes, and the bus answers its publication; nothing here waits for a runtime to have read one.
func deliver(ctx context.Context, d delivery, holder string, sending []readyNode) (map[string]string, error) {
digests := map[string]string{}
if len(sending) == 0 {
return digests, nil
}
send := func(s readyNode) error {
// The number is inside the signed bytes, so a replayed older declaration cannot borrow a
// newer one's (novox/hq 04-ISSUES/107); it was taken when the composition began (issue 204).
body, err := s.declared.Body()
if err != nil {
return err
}
digest, err := d.declare(ctx, s, body)
if err != nil {
return err
}
digests[s.node] = digest
return nil
}
var rest []readyNode
for _, s := range sending {
if holder != "" && s.node == holder {
if err := send(s); err != nil {
return digests, err
}
continue
}
rest = append(rest, s)
}
held := map[string]error{}
if err := d.grant(ctx, sending); err != nil {
var some *grantsRefused
if !errors.As(err, &some) {
var names []string
for _, s := range rest {
names = append(names, s.node)
}
if len(names) == 0 {
return digests, fmt.Errorf("%w: %w", errGrants, err)
}
return digests, fmt.Errorf("%w; %s not sent: %w", errGrants, strings.Join(names, ", "), err)
}
held = some.nodes
}
var notSent []string
for _, s := range rest {
if _, refused := held[s.node]; refused {
notSent = append(notSent, s.node)
continue
}
if err := send(s); err != nil {
return digests, err
}
}
if len(notSent) > 0 {
return digests, fmt.Errorf("%w; %s not sent: %w", errGrants, strings.Join(notSent, ", "),
&grantsRefused{nodes: held})
}
if len(held) > 0 {
// Only the holder's own memberships failed, and it was sent before them.
return digests, fmt.Errorf("%w: %w", errGrants, &grantsRefused{nodes: held})
}
return digests, nil
}
// overTheBus is delivery as the mesh does it: memberships on the bus, declarations signed.
type overTheBus struct {
open *stores
server *link.Server
signer link.Signer
// indent is put before each "sent" line, for the callers whose output is nested.
indent string
}
func (b overTheBus) grant(ctx context.Context, sending []readyNode) error {
return issueMemberships(ctx, b.open, b.server, sending)
}
func (b overTheBus) declare(ctx context.Context, s readyNode, body []byte) (string, error) {
if err := link.Declare(ctx, b.server.Bus(), b.signer, s.node, body, 15*time.Second); err != nil {
return "", err
}
// After it is away, not before. A digest recorded for something that failed to send would make
// the machine look current for a declaration it never received.
digest, err := recordSent(ctx, b.open.inventory, s.node, body)
if err != nil {
return "", err
}
if s.declared.BusUsers != "" {
// And the user list it carried, so the next send reads whether it must go first from the
// list alone (novox/hq issue 249). On the same outliving context as the send's record.
kept, cancel := context.WithTimeout(context.WithoutCancel(ctx), 10*time.Second)
err := b.open.inventory.RecordSentBusUsers(kept, s.node, digestOf([]byte(s.declared.BusUsers)))
cancel()
if err != nil {
return "", err
}
}
fmt.Printf("%ssent %s %d resource(s)\n", b.indent, s.node, len(s.declared.Resources))
return digest, nil
}
// brokerFirst is the machines to send in the order a grant needs (novox/hq issue 249): the machine
// holding the bus first — its declaration carries the bus's user list (composeBusUsers), and a
// module's new permissions are refused by the bus until that list says them. Among the machines it
// is moved to the front; not among them, it is added only when it is behind.
func brokerFirst(names []string, holder string, behind bool) []string {
if holder == "" {
return names
}
present := false
rest := make([]string, 0, len(names))
for _, n := range names {
if n == holder {
present = true
continue
}
rest = append(rest, n)
}
if !present && !behind {
return names
}
return append([]string{holder}, rest...)
}
// brokerBehind is the machine holding the bus — the one whose declaration carries the user list —
// and, when it is not among the machines named, whether the user list it would be sent now differs
// from the one it was last sent (novox/hq issue 249).
//
// **The user list alone, not the whole declaration.** Read from the whole declaration, any change
// pending on that machine — an upgrade its policy records rather than rolls out — went with every
// send anywhere, and a module running there always put it in its first wave. A digest of the list
// last sent is kept for this (ADR 0043: the list is composed on each push, never kept itself).
func brokerBehind(ctx context.Context, open *stores, names []string) (string, bool, error) {
inv := open.inventory
holders, err := seatHolders(ctx, inv)
if err != nil {
return "", false, err
}
h, held := holders[theBrokerSeat]
if !held || h.Node == "" {
return "", false, nil
}
shelf, err := inv.Catalogue(ctx)
if err != nil {
return "", false, err
}
if m, known := shelf[h.Module]; !known || m.BusUsers == "" {
// A holder that is sent no user list carries no grant: nothing to send first.
return "", false, nil
}
for _, n := range names {
if n == h.Node {
return h.Node, false, nil
}
}
plan, _, err := planFor(ctx, open, h.Node)
if err != nil {
// It cannot be worked out: sending it would refuse the whole send, and `plan` says why.
return h.Node, false, nil
}
list, _, err := busUserList(ctx, inv, plan.Modules)
if err != nil {
return h.Node, false, nil
}
sent, err := inv.SentBusUsers(ctx, h.Node)
if err != nil {
return "", false, err
}
return h.Node, userListBehind(list, sent), nil
}
// userListBehind is whether the user list composed now is not the one last sent, by its digest. An
// empty list composed is never behind: there is nothing for it to carry.
func userListBehind(now, sentDigest string) bool {
return now != "" && digestOf([]byte(now)) != sentDigest
}
// couldNotBeResolved is what a push ends with when some machines could not be worked out.
//
// **After the rest have been sent, never instead of sending them.** It is still an error, because
@@ -705,23 +938,36 @@ func couldNotBeResolved(refusals []string, sent int) error {
// consumer and refused on the provider would leave one end holding a credential the other has
// never heard of — which is the state this whole mechanism exists to make impossible.
func sendTo(ctx context.Context, open *stores, names []string) error {
_, err := sendToEach(ctx, open, names)
return err
}
// sendToEach is sendTo, answering the machines it sent: those named, and before them the machine
// holding the bus when its user list must go first (novox/hq issue 249) — so a caller that waits for
// the machines it sent waits for that one too.
func sendToEach(ctx context.Context, open *stores, names []string) ([]string, error) {
inv := open.inventory
ident, err := openIdentity(ctx)
if err != nil {
return err
return nil, err
}
defer ident.Close()
gens, err := generators(ctx, open)
if err != nil {
return err
return nil, err
}
holder, behind, err := brokerBehind(ctx, open, names)
if err != nil {
return nil, err
}
names = brokerFirst(names, holder, behind)
// Held from composing to sending (novox/hq ADR 0100); a caller that holds them already —
// converge, which flips the node and then sends it — is not made to wait on itself.
ctx, release, err := holdNodes(ctx, open, names)
if err != nil {
return err
return nil, err
}
defer release()
@@ -750,33 +996,29 @@ func sendTo(ctx context.Context, open *stores, names []string) error {
sending = append(sending, readyNode{name, declared})
}
if len(refusals) > 0 {
return fmt.Errorf("nothing was sent. %d machine(s) could not be resolved:\n\n%s",
return nil, fmt.Errorf("nothing was sent. %d machine(s) could not be resolved:\n\n%s",
len(refusals), strings.Join(refusals, "\n\n"))
}
server, err := connectLink(ctx, nil, nil, nil)
if err != nil {
return err
return nil, err
}
defer server.Close()
for _, s := range sending {
body, err := s.declared.Body()
if err != nil {
return err
}
if err := link.Declare(ctx, server.Bus(), ident, s.node, body, 15*time.Second); err != nil {
return err
}
if _, err := recordSent(ctx, inv, s.node, body); err != nil {
return err
}
fmt.Printf(" sent %s %d resource(s)\n", s.node, len(s.declared.Resources))
}
// And every assignment on those machines its membership (novox/hq ADR 0160): composed from the
// same records the bus's accounts are, so what a runtime serves and what its account may are one
// composition. Issued after the declaration, because the runtime it is for arrives with it.
return issueMemberships(ctx, open, server, sending)
// composition. **Issued before the declarations** (novox/hq issue 249): the runtime the
// membership is for arrives with the declaration, and a membership waits for it on the bus; the
// code arriving first was refused its own state until somebody pushed.
if _, err := deliver(ctx, overTheBus{open: open, server: server, signer: ident, indent: " "}, holder, sending); err != nil {
return nil, err
}
sent := make([]string, 0, len(sending))
for _, s := range sending {
sent = append(sent, s.node)
}
return sent, nil
}
// issueMemberships publishes the membership of every module on the machines just sent.
@@ -797,18 +1039,23 @@ func issueMemberships(ctx context.Context, open *stores, server *link.Server, se
// **Every declared state's bucket, before the memberships that name it** (novox/hq ADR 0201). The
// raise at start asserts them too, but a module registered and assigned since would otherwise have
// its bucket only after the control plane next restarts — found the first time a module declared
// state: its bundle asked for a bucket that did not exist. Idempotent and cheap; a failure is said
// and the push stands, as a membership's is.
if buckets, err := open.inventory.DeclaredBuckets(ctx); err != nil {
fmt.Printf(" the modules' state could not be read, so no bucket was asserted: %v\n", err)
} else if _, err := broker.RaiseBuckets(broker.OnConn(bus.Conn), buckets); err != nil {
fmt.Printf(" the modules' state could not be asserted on the bus: %v — the next push tries again\n", err)
// state: its bundle asked for a bucket that did not exist. Idempotent and cheap.
//
// **A failure here is the send's failure** (novox/hq issue 249). It was said and the push stood,
// because the declarations were already away; they are sent after this now — all but the bus's
// own machine, sent before it (deliver) — and a module whose state does not exist is a module
// that fails its start, so they are not sent and the caller tries again rather than reporting the
// rollout done.
buckets, err := open.inventory.DeclaredBuckets(ctx)
if err != nil {
return fmt.Errorf("the modules' state could not be read, so no bucket was asserted: %w", err)
}
// The declarations are sent and recorded by now; a membership that cannot be issued is said
// and does not unsay them. Every runtime without one serves the shape it derives (ADR 0160), so
// the push stands, the first failure is named once, and the next push tries again.
issued, failed := 0, 0
var first error
if _, err := broker.RaiseBuckets(broker.OnConn(bus.Conn), buckets); err != nil {
return fmt.Errorf("the modules' state could not be asserted on the bus: %w", err)
}
// Every membership is tried, and the first failure named once.
issued := 0
refused := map[string]error{}
for _, s := range sent {
node := s.node
for _, d := range records.Assigned[node] {
@@ -829,10 +1076,9 @@ func issueMemberships(ctx context.Context, open *stores, server *link.Server, se
return err
}
if err := bus.PublishMembership(ctx, node, d.Module, body); err != nil {
if first == nil {
first = err
if refused[node] == nil {
refused[node] = fmt.Errorf("%s: %w", d.Module, err)
}
failed++
continue
}
issued++
@@ -841,9 +1087,10 @@ func issueMemberships(ctx context.Context, open *stores, server *link.Server, se
if issued > 0 {
fmt.Printf(" issued %d membership(s)\n", issued)
}
if failed > 0 {
fmt.Printf(" %d membership(s) could not be issued; the first: %v — the machines keep what "+
"they derive until the next push\n", failed, first)
if len(refused) > 0 {
// Returned, never passed over (novox/hq issue 249): the declarations of the machines they are
// for are not sent, and the rollout that asked is tried again rather than waiting for a push.
return &grantsRefused{nodes: refused}
}
return nil
}
+292 -26
View File
@@ -184,6 +184,7 @@ func planOfMerge(m link.SourceMoved, moved []string, edges []inventory.Edge) inv
return inventory.Plan{
ID: fmt.Sprintf("plan-%d", time.Now().UnixNano()),
Repository: m.Owner + "/" + m.Repo,
Branch: m.Base,
Commit: m.Commit,
Created: time.Now().UTC(),
State: inventory.PlanBuilding,
@@ -192,6 +193,57 @@ func planOfMerge(m link.SourceMoved, moved []string, edges []inventory.Edge) inv
}
}
// supersededBy is what a newer plan takes over from the open plans it supersedes (novox/hq issue
// 254, ADR 0218): the modules they had not finished, and those plans closed as superseded.
//
// **A merge looked at no plan but its own.** Two merges of one repository a few minutes apart were
// two open plans asking for the same modules, each sending machines what it built; and a plan that
// would never move again — waiting on a report that could not come, at 97b1b2b — stayed open for
// ever beside the newer ones, read as work in progress by everyone who looked. The newer merge is the
// newer intent for that repository and branch, so its plan takes over: every open plan of the same
// repository and branch **created before it** — by the time the plans were made, never by comparing
// commits, which have no order of their own — gives up the modules it had not built, and those are
// planned again in the newer plan beside what the newer merge moved.
//
// "Not built" is a module not yet asked, or asked and not answered; **and a module built and not
// yet sent to its machines**, where its policy rolls it out: closed, the older plan would never send
// it, and the catalogue announces no move for a rebuild (issue 189), so the newer plan builds and
// sends it. A build the older plan asked still finishes and registers as any build does — ordered by
// when it was asked (issue 219), so the newer plan's ask, made later, is the one that stands.
//
// A plan with no branch recorded is from before branches were kept, and is superseded by the next
// plan of its repository: what it had not built is folded in, so nothing is lost by it.
func supersededBy(newer inventory.Plan, open []inventory.Plan, rollsOut func(string) bool) ([]string, []inventory.Plan) {
folded := map[string]bool{}
var closed []inventory.Plan
for _, old := range open {
if old.ID == newer.ID || !old.Open() || !strings.EqualFold(old.Repository, newer.Repository) ||
(old.Branch != "" && old.Branch != newer.Branch) || !old.Created.Before(newer.Created) {
continue
}
var took []string
for name, s := range old.Modules {
if s == nil || s.State != "built" || (s.SentAt == nil && rollsOut(name)) {
folded[name] = true
took = append(took, name)
}
}
sort.Strings(took)
old.State = inventory.PlanSuperseded
old.Note = fmt.Sprintf("superseded at tier %d by %s (%s at %s)", old.Tier, newer.ID, newer.Repository, short(newer.Commit))
if len(took) > 0 {
old.Note += "; " + strings.Join(took, ", ") + " planned there again"
}
closed = append(closed, old)
}
out := make([]string, 0, len(folded))
for name := range folded {
out = append(out, name)
}
sort.Strings(out)
return out, closed
}
// gates is what the next tier needs running from this one: a module of the tier that a later
// tier is built by — the runtime dependency — and whose policy rolls it out, must be applied by
// the machines running it before the next tier is asked. A base an image stands on need only be
@@ -230,6 +282,22 @@ func gates(p inventory.Plan, edges []inventory.Edge, rollsOut func(string) bool)
}
// applied says whether every machine running the module has reported since the module was built.
// appliedEach is applied with a moment of its own for each machine: the reports that count are the ones
// after that machine was sent the build (novox/hq issue 256).
func appliedEach(module string, since func(node string) time.Time, running []string, reports []inventory.Reported) (bool, []string) {
at := map[string]*time.Time{}
for _, r := range reports {
at[r.Node] = r.At
}
var waiting []string
for _, n := range running {
if t := at[n]; t == nil || t.Before(since(n)) {
waiting = append(waiting, n)
}
}
return len(waiting) == 0, waiting
}
func applied(module string, builtAt time.Time, running []string, reports []inventory.Reported) (bool, []string) {
at := map[string]*time.Time{}
for _, r := range reports {
@@ -339,6 +407,10 @@ func planBuilt(ctx context.Context, open *stores, module, commit, failed string,
state.Why = failed
p.State = inventory.PlanFailed
p.Note = fmt.Sprintf("%s failed to build in tier %d", module, p.Tier)
sayUnsent(p, func(m string) bool {
u, err := inv.UpgradeOf(ctx, m)
return err == nil && u.RollOut
})
} else {
state.State = "built"
state.BuiltAt = &now
@@ -400,8 +472,17 @@ func advanceHeld(ctx context.Context, open *stores) {
moved, err := advanceOnce(ctx, open, p, edges, rollsOut)
if err != nil {
fmt.Printf("%s: %v\n", p.ID, err)
// Kept in the plan, so `plans` says why it has not moved rather than the log alone;
// the state is left as it was and the step is tried again on the next tick.
p.Note = "tier " + fmt.Sprint(p.Tier) + ": " + err.Error() + " — tried again"
if err := inv.SavePlan(ctx, *p); err != nil {
fmt.Printf("%s: cannot keep the plan: %v\n", p.ID, err)
}
break
}
if p.State == inventory.PlanFailed {
sayUnsent(p, rollsOut)
}
if err := inv.SavePlan(ctx, *p); err != nil {
fmt.Printf("%s: cannot keep the plan: %v\n", p.ID, err)
break
@@ -470,6 +551,15 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan,
// a module that packages another repository's source, keeps its commit; the catalogue announces
// no move for it and its machines would keep the old image until somebody pushed (novox/hq
// issue 189). A module whose policy records is built and left, as its policy says.
//
// **One machine first, unless the module's policy says together** (novox/hq issue 249, ADR
// 0218). The plan sent every machine running the module at once, and the operator's policy —
// one at a time, stopping at the first that fails, which an announced upgrade honours — was not
// read here at all: a module whose new declarations broke it broke everywhere in the same
// minute. Now the first machine is sent, the plan records it and waits for that machine's report
// after the send to say it applied what it was sent; only then are the rest sent. A first machine
// that fails or refuses stops the module's rollout and the plan with it, the rest untouched.
var pending []string
for _, m := range tier {
state := p.Modules[m]
if state == nil || state.SentAt != nil || !rollsOut(m) {
@@ -479,17 +569,64 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan,
if err != nil {
return false, err
}
policy, err := inv.UpgradeOf(ctx, m)
if err != nil {
return false, err
}
var reports []inventory.Reported
if !policy.Together {
// Read for the choice of the first machine as well as for its report.
if reports, err = inv.LastReports(ctx); err != nil {
return false, err
}
}
now := time.Now().UTC()
state.SentAt = &now
if len(running) == 0 {
step := nextRollout(*state, running, policy.Together, reports, now, planWaitBound)
switch {
case step.failed != "":
state.Why = step.failed
p.State = inventory.PlanFailed
p.Note = fmt.Sprintf("%s stopped at its first machine in tier %d: %s; %s left as it was",
m, p.Tier, step.failed, orNone(strings.Join(step.rest, ", ")))
fmt.Printf("%s: %s\n", p.ID, p.Note)
return true, nil
case step.waiting != "":
pending = append(pending, fmt.Sprintf("%s on %s, sent first at %s", m, step.waiting, state.FirstAt.Local().Format("15:04")))
continue
case len(step.send) == 0:
// No machine runs it: nothing to send, and nothing to wait for.
state.SentAt = &now
continue
}
if err := sendTo(ctx, open, running); err != nil {
return false, fmt.Errorf("sending %s to %s after tier %d: %w", m, strings.Join(running, ", "), p.Tier, err)
// What sendToEach answers, not what was asked: the machine holding the bus is sent before
// the first when its user list must change (issue 249), and the plan waits for it too.
sent, err := sendToEach(ctx, open, step.send)
if err != nil {
// Not marked sent, so the next step tries again (issue 249): a grant that could not be
// issued is a send that did not happen.
return false, fmt.Errorf("sending %s to %s after tier %d: %w", m, strings.Join(step.send, ", "), p.Tier, err)
}
fmt.Printf("%s: tier %d built; sent %s to %s\n", p.ID, p.Tier, m, strings.Join(running, ", "))
if step.first {
state.First = sent
state.FirstAt = &now
p.State = inventory.PlanRolling
p.Note = fmt.Sprintf("tier %d built; sent %s to %s first", p.Tier, m, strings.Join(sent, ", "))
fmt.Printf("%s: tier %d built; sent %s to %s first, the rest once it reports it applied\n",
p.ID, p.Tier, m, strings.Join(sent, ", "))
return true, nil
}
state.SentAt = &now
fmt.Printf("%s: tier %d built; sent %s to %s\n", p.ID, p.Tier, m, strings.Join(sent, ", "))
return true, nil
}
if len(pending) > 0 {
note := "tier " + fmt.Sprint(p.Tier) + " built; waiting for " + strings.Join(pending, "; ") +
" to report it applied before the rest are sent"
changed := p.State != inventory.PlanRolling || p.Note != note
p.State = inventory.PlanRolling
p.Note = note
return changed, nil
}
// And wait for what the next tier needs running.
needed := gates(*p, edges, rollsOut)
if len(needed) > 0 {
@@ -516,10 +653,24 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan,
if state.BuiltAt != nil {
since = *state.BuiltAt
}
if state.SentAt != nil && state.SentAt.After(since) {
since = *state.SentAt
// **Each machine from its own send** (novox/hq issue 256). With one machine first (ADR 0218)
// a module is sent twice — the first machine, then the rest — and SentAt is the second.
// Asked of every machine, the first machine's report, made between the two sends, read as
// older than the build, and the gate waited for a report it already had, for ever.
sinceFor := func(node string) time.Time {
at := since
sent := state.SentAt
for _, n := range state.First {
if n == node {
sent = state.FirstAt
}
}
if sent != nil && sent.After(at) {
at = *sent
}
return at
}
if ok, on := applied(m, since, running, reports); !ok {
if ok, on := appliedEach(m, sinceFor, running, reports); !ok {
waiting = append(waiting, fmt.Sprintf("%s on %s", m, strings.Join(on, ", ")))
}
}
@@ -540,6 +691,110 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan,
return true, nil
}
// rolloutStep is what a plan does next with one built module's machines (novox/hq issue 249).
type rolloutStep struct {
// send is the machines to send now; first, whether they are the first machine's send.
send []string
first bool
// waiting names the first machines whose report the rest wait for.
waiting string
// failed says how a first machine did not take it; rest is what is then left alone.
failed string
rest []string
}
// nextRollout is the next step of one module's rollout in a plan (novox/hq issue 249, ADR 0218).
//
// Together, every machine running it at once, as the policy says. Otherwise one machine first — the
// first by name among those that have reported within the bound, so a laptop that is away is not
// the one the rest wait on; the first by name when none has; the same choice on every controller and
// every resume — and the rest once each machine the first send reached reports, about the
// declaration it was last sent, that it applied it. Compared by the store's own record of what was
// sent (Reported.Current), never by this controller's clock against the machine's.
//
// **A first machine that fails, refuses, or does not report within the bound stops the rollout
// there** (ADR 0218 §2), naming the machine; the rest are not sent. A wait with no end is not a
// rollout: it held the plan open for ever, read as work in progress (issue 254).
func nextRollout(s inventory.PlanModule, running []string, together bool, reports []inventory.Reported,
now time.Time, bound time.Duration) rolloutStep {
if len(running) == 0 {
return rolloutStep{}
}
if together {
return rolloutStep{send: running}
}
byNode := map[string]inventory.Reported{}
for _, r := range reports {
byNode[r.Node] = r
}
if s.FirstAt == nil {
sorted := append([]string{}, running...)
sort.Strings(sorted)
for _, n := range sorted {
if r, said := byNode[n]; said && r.At != nil && now.Sub(*r.At) <= bound {
return rolloutStep{send: []string{n}, first: true}
}
}
return rolloutStep{send: sorted[:1], first: true}
}
sentFirst := map[string]bool{}
for _, n := range s.First {
sentFirst[n] = true
}
var rest []string
for _, n := range running {
if !sentFirst[n] {
rest = append(rest, n)
}
}
var waiting, failed []string
for _, n := range s.First {
r, said := byNode[n]
// Only a report about what it was last sent says anything about this build.
if !said || r.At == nil || !r.Current {
waiting = append(waiting, n)
continue
}
switch r.Outcome {
case inventory.OutcomeApplied:
case inventory.OutcomeFailed, inventory.OutcomeRefused:
failed = append(failed, n+" "+r.Outcome+" what it was sent")
default:
waiting = append(waiting, n)
}
}
if len(failed) > 0 {
return rolloutStep{failed: strings.Join(failed, "; "), rest: rest}
}
if len(waiting) > 0 {
if now.Sub(*s.FirstAt) > bound {
return rolloutStep{failed: fmt.Sprintf("%s did not report it applied within %s",
strings.Join(waiting, ", "), bound), rest: rest}
}
return rolloutStep{waiting: strings.Join(waiting, ", ")}
}
return rolloutStep{send: rest}
}
// sayUnsent adds to an ended plan's note the modules it built and never sent (novox/hq issue 249).
// An announced move of a module a plan held was left to that plan; a plan that ends without
// sending it — failed elsewhere, or closed by hand — would leave its machines behind with nothing
// saying so. A module whose rollout stopped at its first machine is not among them: that stop was
// the point. Said once.
func sayUnsent(p *inventory.Plan, rollsOut func(string) bool) {
var unsent []string
for name, s := range p.Modules {
if s != nil && s.State == "built" && s.SentAt == nil && s.FirstAt == nil && rollsOut(name) {
unsent = append(unsent, name)
}
}
if len(unsent) == 0 || strings.Contains(p.Note, "built and never sent") {
return
}
sort.Strings(unsent)
p.Note += "; built and never sent: " + strings.Join(unsent, ", ") + " — `push --behind` sends them"
}
// planTicker advances open plans on a timer, for the steps outcomes alone cannot take.
func planTicker(ctx context.Context, open *stores) {
advancePlans(ctx, open)
@@ -563,8 +818,10 @@ func planLine(p inventory.Plan, now time.Time) string {
return fmt.Sprintf("%s %s done, %d tier(s)", p.Repository, short(p.Commit), len(p.Tiers))
case inventory.PlanFailed:
return fmt.Sprintf("%s %s FAILED at %s: %s", p.Repository, short(p.Commit), where, p.Note)
case inventory.PlanSuperseded:
return fmt.Sprintf("%s %s %s", p.Repository, short(p.Commit), p.Note)
}
since := now.Sub(p.Updated).Round(time.Minute)
since := now.Sub(p.Updated).Round(time.Second)
late := ""
if since > planWaitBound {
late = " — LATE"
@@ -693,7 +950,19 @@ func plansCommand(ctx context.Context, args []string) error {
if *whatIf != "" {
return planWhatIf(ctx, inv, *whatIf, splitList(*paths), splitList(*modules))
}
if len(positionals) == 2 && positionals[0] == "stop" {
// `stop`, or `close` (novox/hq issue 254): a person ending a plan that will not move again — one
// waiting on a report that cannot come — so it stops reading as work in progress. Marked failed
// with who ended it; what it asked still builds and registers.
if len(positionals) == 2 && (positionals[0] == "stop" || positionals[0] == "close") {
how := "stopped"
if positionals[0] == "close" {
how = "closed"
}
release, err := inv.HoldPlans(ctx, true)
if err != nil {
return err
}
defer release()
p, err := inv.PlanByID(ctx, positionals[1])
if err != nil {
return err
@@ -702,25 +971,16 @@ func plansCommand(ctx context.Context, args []string) error {
return fmt.Errorf("%s is already %s", p.ID, p.State)
}
p.State = inventory.PlanFailed
p.Note = "stopped by hand at tier " + fmt.Sprint(p.Tier)
release, err := inv.HoldPlans(ctx, true)
if err != nil {
return err
}
defer release()
if p, err = inv.PlanByID(ctx, positionals[1]); err != nil {
return err
}
if !p.Open() {
return fmt.Errorf("%s is already %s", p.ID, p.State)
}
p.State = inventory.PlanFailed
p.Note = "stopped by hand at tier " + fmt.Sprint(p.Tier)
p.Note = how + " by hand at tier " + fmt.Sprint(p.Tier)
sayUnsent(&p, func(m string) bool {
u, err := inv.UpgradeOf(ctx, m)
return err == nil && u.RollOut
})
if err := inv.SavePlan(ctx, p); err != nil {
return err
}
fmt.Printf("%s stopped at tier %d of %d; what was asked still builds and registers, nothing further is asked\n",
p.ID, p.Tier, len(p.Tiers))
fmt.Printf("%s %s at tier %d of %d; what was asked still builds and registers, nothing further is asked\n",
p.ID, how, p.Tier, len(p.Tiers))
return nil
}
plans, err := inv.RecentPlans(ctx, *limit)
@@ -795,7 +1055,13 @@ func planWhatIf(ctx context.Context, inv *inventory.Inventory, repository string
how := "built; its policy records, so nothing is sent"
if u, err := inv.UpgradeOf(ctx, name); err == nil && u.RollOut {
running, _ := inv.Running(ctx, name)
reports, _ := inv.LastReports(ctx)
how = "built, then sent to " + orNone(strings.Join(running, ", "))
// One machine first unless the policy says together (novox/hq issue 249).
if first := nextRollout(inventory.PlanModule{}, running, u.Together, reports, time.Now(), planWaitBound); first.first && len(running) > 1 {
how = fmt.Sprintf("built, then sent to %s first and to the rest once it has applied it",
first.send[0])
}
rolls[name] = how
}
fmt.Printf(" %-22s %s\n", name, how)
+31
View File
@@ -181,3 +181,34 @@ func TestAPlanSettlesAnAskedBuildFromTheRecords(t *testing.T) {
t.Errorf("a recorded failure did not fail the plan: %+v %+v", q, q.Modules["x"])
}
}
// The first machine's report, made between the send to it and the send to the rest, opens the gate for it:
// each machine is judged from its own send, not from the last one (novox/hq issue 256).
func TestTheGateJudgesEachMachineFromItsOwnSend(t *testing.T) {
built := time.Date(2026, 10, 5, 18, 22, 0, 0, time.UTC)
firstSent := built.Add(31 * time.Second)
firstReported := built.Add(43 * time.Second)
restSent := built.Add(58 * time.Second)
restReported := built.Add(74 * time.Second)
reports := []inventory.Reported{
{Node: "ace", At: &firstReported},
{Node: "g14", At: &restReported},
}
since := func(node string) time.Time {
if node == "ace" {
return firstSent
}
return restSent
}
if ok, waiting := appliedEach("build-agent", since, []string{"ace", "g14"}, reports); !ok {
t.Fatalf("the gate still waits on %v, though each reported after its own send", waiting)
}
// The old reading, every machine from the last send, is what held the plan.
if ok, _ := applied("build-agent", restSent, []string{"ace", "g14"}, reports); ok {
t.Fatal("the single-moment reading should hold the first machine back")
}
early := built.Add(10 * time.Second)
if ok, waiting := appliedEach("build-agent", since, []string{"ace"}, []inventory.Reported{{Node: "ace", At: &early}}); ok || waiting[0] != "ace" {
t.Fatal("a report from before the machine was sent opened the gate")
}
}
+152
View File
@@ -0,0 +1,152 @@
package main
import (
"reflect"
"strings"
"testing"
"time"
"github.com/novox/mesh-controller/internal/inventory"
)
// novox/hq issue 249, ADR 0218: a plan rolls a module out to one machine first and the rest only
// once that machine has reported it applied; a module whose policy says together goes everywhere at
// once, as before.
func TestAPlanSendsOneMachineFirstAndTheRestAfterItsReport(t *testing.T) {
running := []string{"novox", "ace", "g14"}
sentAt := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC)
now := sentAt.Add(5 * time.Minute)
bound := 30 * time.Minute
after := sentAt.Add(time.Minute)
next := func(s inventory.PlanModule, running []string, together bool, reports []inventory.Reported) rolloutStep {
return nextRollout(s, running, together, reports, now, bound)
}
// Together: every machine at once.
if step := next(inventory.PlanModule{}, running, true, nil); !reflect.DeepEqual(step.send, running) || step.first {
t.Fatalf("a together policy did not send every machine at once: %+v", step)
}
// Otherwise the first by name, alone, when none has reported lately.
step := next(inventory.PlanModule{}, running, false, nil)
if !step.first || !reflect.DeepEqual(step.send, []string{"ace"}) {
t.Fatalf("the first send was %+v, wanted ace alone", step)
}
state := inventory.PlanModule{First: []string{"ace"}, FirstAt: &sentAt}
report := func(outcome string, current bool) []inventory.Reported {
return []inventory.Reported{{Node: "ace", At: &after, Outcome: outcome, Current: current},
{Node: "g14", At: &after, Outcome: inventory.OutcomeApplied, Current: true}}
}
// No report yet, or one about an older declaration than it was last sent: wait.
for what, reports := range map[string][]inventory.Reported{
"no report": nil,
"a report about older": report(inventory.OutcomeApplied, false),
} {
step := next(state, running, false, reports)
if len(step.send) != 0 || step.waiting != "ace" || step.failed != "" {
t.Errorf("%s: %+v, wanted to wait for ace", what, step)
}
}
// Applied what it was last sent: the rest, and only the rest.
step = next(state, running, false, report(inventory.OutcomeApplied, true))
if step.first || !reflect.DeepEqual(step.send, []string{"novox", "g14"}) {
t.Fatalf("after ace applied it the plan sent %+v, wanted novox and g14", step)
}
// Failed or refused: stop, the rest untouched.
for _, outcome := range []string{inventory.OutcomeFailed, inventory.OutcomeRefused} {
step := next(state, running, false, report(outcome, true))
if len(step.send) != 0 || !strings.Contains(step.failed, "ace "+outcome) ||
!reflect.DeepEqual(step.rest, []string{"novox", "g14"}) {
t.Errorf("a first machine that %s it: %+v", outcome, step)
}
}
// The machine holding the bus went with the first send: the rest wait for it too, and it is
// not sent again.
both := inventory.PlanModule{First: []string{"novox", "ace"}, FirstAt: &sentAt}
half := report(inventory.OutcomeApplied, true)
if step := next(both, running, false, half); step.waiting != "novox" {
t.Fatalf("the plan did not wait for the bus's machine sent first: %+v", step)
}
all := append(half, inventory.Reported{Node: "novox", At: &after, Outcome: inventory.OutcomeApplied, Current: true})
if step := next(both, running, false, all); !reflect.DeepEqual(step.send, []string{"g14"}) {
t.Fatalf("after both applied it the plan sent %+v, wanted g14 alone", step)
}
// One machine, or none: nothing is waited for that cannot come.
if step := next(inventory.PlanModule{}, nil, false, nil); len(step.send) != 0 || step.first {
t.Fatalf("a module nothing runs was sent: %+v", step)
}
if step := next(state, []string{"ace"}, false, report(inventory.OutcomeApplied, true)); len(step.send) != 0 || step.waiting != "" {
t.Fatalf("a module on one machine waited for more: %+v", step)
}
}
// ADR 0218 §2: a first machine that does not report within the bound stops the rollout there,
// naming the machine and the bound; the rest are left alone.
func TestAFirstMachineThatDoesNotReportStopsTheRollout(t *testing.T) {
sentAt := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC)
state := inventory.PlanModule{First: []string{"ace"}, FirstAt: &sentAt}
step := nextRollout(state, []string{"ace", "g14"}, false, nil, sentAt.Add(31*time.Minute), 30*time.Minute)
if !strings.Contains(step.failed, "ace did not report it applied within 30m") ||
!reflect.DeepEqual(step.rest, []string{"g14"}) || len(step.send) != 0 {
t.Fatalf("a silent first machine: %+v", step)
}
}
// The first machine is the first by name among those heard from lately: a laptop that is away is
// not the one the rest wait on. When none has been heard from, the first by name.
func TestTheFirstMachineIsOneThatHasReportedLately(t *testing.T) {
now := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC)
lately, long := now.Add(-time.Minute), now.Add(-3*time.Hour)
reports := []inventory.Reported{
{Node: "ace", At: &long}, {Node: "g14", At: &lately}, {Node: "novox", At: &lately},
}
step := nextRollout(inventory.PlanModule{}, []string{"novox", "ace", "g14"}, false, reports, now, 30*time.Minute)
if !step.first || !reflect.DeepEqual(step.send, []string{"g14"}) {
t.Fatalf("the first send was %+v, wanted g14, the first heard from lately", step)
}
}
// An announced move of a module an open plan is still rolling out is left to the plan: sending it
// here as well put the bundle on every machine at once (novox/hq issue 249).
func TestAnAnnouncedMoveIsLeftToThePlanRollingItOut(t *testing.T) {
sent := time.Now()
plans := []inventory.Plan{
{ID: "plan-done", State: inventory.PlanDone, Modules: map[string]*inventory.PlanModule{"agent": {}}},
{ID: "plan-1", State: inventory.PlanRolling, Modules: map[string]*inventory.PlanModule{
"agent": {State: "built", First: []string{"ace"}, FirstAt: &sent}, "gitea": {State: "built", SentAt: &sent}}},
}
if got := rolledOutByAPlan(plans, "agent"); got != "plan-1" {
t.Fatalf("a module the plan is rolling out was not left to it: %q", got)
}
for _, m := range []string{"gitea", "keycloak"} {
if got := rolledOutByAPlan(plans, m); got != "" {
t.Errorf("%s, which no plan will send, was left to %s", m, got)
}
}
}
// A plan that ends without sending what it built says so, with the remedy; a module whose rollout
// stopped at its first machine is not among them.
func TestAnEndedPlanSaysWhatItBuiltAndNeverSent(t *testing.T) {
at := time.Now()
p := inventory.Plan{State: inventory.PlanFailed, Note: "closed by hand at tier 1",
Modules: map[string]*inventory.PlanModule{
"agent": {State: "built"},
"stopped": {State: "built", First: []string{"ace"}, FirstAt: &at},
"sent": {State: "built", SentAt: &at},
"notes": {State: "built"},
"later": {},
}}
rollsOut := func(m string) bool { return m != "notes" }
sayUnsent(&p, rollsOut)
sayUnsent(&p, rollsOut)
if p.Note != "closed by hand at tier 1; built and never sent: agent — `push --behind` sends them" {
t.Fatalf("the note reads %q", p.Note)
}
}
+4 -1
View File
@@ -100,6 +100,9 @@ func argvFor(verb string, args map[string]any) ([]string, error) {
if id := str("stop"); id != "" {
return []string{"plans", "stop", id}, nil
}
if id := str("close"); id != "" {
return []string{"plans", "close", id}, nil
}
if id := str("id"); id != "" {
return []string{"plans", id}, nil
}
@@ -214,7 +217,7 @@ func argvFor(verb string, args map[string]any) ([]string, error) {
}
// jsonVerbs are the verbs whose command speaks JSON, so the answer carries it as data as well.
var jsonVerbs = map[string]bool{"status": true, "seats": true, "plan": true}
var jsonVerbs = map[string]bool{"status": true, "seats": true, "plan": true, "collection": true}
// runVerb runs this binary with the given command line and gathers what it said.
func runVerb(ctx context.Context, argv []string) (verbAnswer, error) {
+4
View File
@@ -31,6 +31,10 @@ type sendable struct {
// the same composition as its received files, and every machine's private-network address.
Received map[string]map[string][]catalogue.Contribution
Mesh []string
// BusUsers is the bus's user list this declaration carries, empty for every machine but the one
// holding the bus; not sent apart from the file it is in. Its digest is recorded once sent, so
// whether that machine must go first is read from the list alone (novox/hq issue 249).
BusUsers string
// LeftOut is every module of the machine's set left out of this declaration because a stored
// setting cannot compose with its definition (novox/hq ADR 0163, rule 6), sorted. The host
// keeps that module's held things and touches none of its containers; a machine is told
+95
View File
@@ -0,0 +1,95 @@
package main
import (
"reflect"
"strings"
"testing"
"time"
"github.com/novox/mesh-controller/internal/inventory"
)
// novox/hq issue 254, ADR 0218: a newer plan takes over what the older open plans of its repository
// and branch had not built, and closes them as superseded; another repository's plan, another
// branch's, and a plan made after it are left alone.
func TestANewerPlanSupersedesTheOlderOpenPlansOfItsRepository(t *testing.T) {
at := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC)
sent := at.Add(time.Minute)
plan := func(id, repository, branch string, created time.Time, modules map[string]*inventory.PlanModule) inventory.Plan {
return inventory.Plan{ID: id, Repository: repository, Branch: branch, Commit: id + "-commit",
Created: created, State: inventory.PlanRolling, Modules: modules}
}
older := plan("plan-1", "novox/mesh-catalog", "main", at, map[string]*inventory.PlanModule{
"gitea": {State: "built", SentAt: &sent}, // done with: stays done
"keycloak": {State: "asked"}, // asked, not answered: folded
"plex": {}, // not yet asked: folded
"agent": {State: "built"}, // built, rolls out, not sent: folded
"notes": {State: "built"}, // built, records: nothing to send
})
stuck := plan("plan-0", "Novox/Mesh-Catalog", "", at.Add(-time.Hour), map[string]*inventory.PlanModule{
"runtime": {State: "asked"},
})
other := plan("plan-2", "novox/mesh-controller", "main", at, map[string]*inventory.PlanModule{"mesh-controller": {}})
release := plan("plan-3", "novox/mesh-catalog", "release", at, map[string]*inventory.PlanModule{"lemurs": {}})
later := plan("plan-5", "novox/mesh-catalog", "main", at.Add(2*time.Hour), map[string]*inventory.PlanModule{"later": {}})
done := plan("plan-6", "novox/mesh-catalog", "main", at, map[string]*inventory.PlanModule{"finished": {}})
done.State = inventory.PlanDone
newer := plan("plan-4", "novox/mesh-catalog", "main", at.Add(time.Hour), nil)
newer.Commit = "97b1b2b0c0ffee"
rollsOut := func(m string) bool { return m != "notes" }
folded, closed := supersededBy(newer, []inventory.Plan{stuck, older, other, release, later, done, newer}, rollsOut)
if want := []string{"agent", "keycloak", "plex", "runtime"}; !reflect.DeepEqual(folded, want) {
t.Fatalf("folded %v, wanted %v", folded, want)
}
var ids []string
for _, p := range closed {
ids = append(ids, p.ID)
if p.State != inventory.PlanSuperseded || p.Open() {
t.Errorf("%s was left %s", p.ID, p.State)
}
if !strings.Contains(p.Note, "plan-4") || !strings.Contains(p.Note, "97b1b2b0") {
t.Errorf("%s does not name the plan that superseded it: %q", p.ID, p.Note)
}
}
if want := []string{"plan-0", "plan-1"}; !reflect.DeepEqual(ids, want) {
t.Fatalf("superseded %v, wanted %v — another repository, another branch, a later plan and a "+
"finished one are left alone", ids, want)
}
if other.State != inventory.PlanRolling {
t.Fatal("the plan handed in was changed in place")
}
if line := planLine(closed[1], time.Now()); !strings.Contains(line, "superseded") {
t.Fatalf("a superseded plan reads %q", line)
}
}
// novox/hq issue 254: a person closes a plan that will not move again, by its id.
func TestAPersonClosesAStuckPlan(t *testing.T) {
open := aMesh(t)
ctx := t.Context()
stuck := inventory.Plan{ID: "plan-97b1b2b", Repository: "novox/mesh-catalog", Commit: "97b1b2b",
Created: time.Now().UTC(), State: inventory.PlanRolling, Tier: 1, Tiers: [][]string{{"a"}, {"b"}},
Modules: map[string]*inventory.PlanModule{"a": {State: "built"}, "b": {}}}
if err := open.inventory.SavePlan(ctx, stuck); err != nil {
t.Fatal(err)
}
if err := plansCommand(ctx, []string{"close", stuck.ID}); err != nil {
t.Fatal(err)
}
closed, err := open.inventory.PlanByID(ctx, stuck.ID)
if err != nil {
t.Fatal(err)
}
if closed.State != inventory.PlanFailed || !strings.Contains(closed.Note, "closed by hand") {
t.Fatalf("the plan was left %s: %q", closed.State, closed.Note)
}
if err := plansCommand(ctx, []string{"close", stuck.ID}); err == nil {
t.Fatal("a plan already closed was closed again")
}
if argv, err := argvFor("plans", map[string]any{"close": stuck.ID}); err != nil ||
!reflect.DeepEqual(argv, []string{"plans", "close", stuck.ID}) {
t.Fatalf("the seat's verb does not close a plan: %v %v", argv, err)
}
}
+137 -10
View File
@@ -5,6 +5,7 @@ import (
"errors"
"flag"
"fmt"
"path"
"regexp"
"strings"
"time"
@@ -55,10 +56,26 @@ func (f following) Upgraded(ctx context.Context, u link.Upgraded) error {
return nil
}
// **A plan that holds the module rolls it out, and this does not** (novox/hq issue 249, ADR
// 0218). A merge's plan builds the module and sends it one machine first, the rest once that one
// has applied it; this announcement arrives as the build registers, and sending here too — one
// machine after another without waiting for any to apply — put the new bundle on every machine in
// the same minute, whatever the plan was waiting for. A move no plan answers (a build asked by
// hand) is still this handler's.
if plans, err := inv.OpenPlans(ctx); err != nil {
return notNow(err)
} else if id := rolledOutByAPlan(plans, u.Module); id != "" {
// Said with its remedy: a plan that ends without sending it — failed, or closed by hand — leaves
// these machines behind, which `status` lists and `push --behind` sends (novox/hq issue 249).
fmt.Printf("%s moved to %s; %s rolls it out to %s — if that plan ends without sending it, "+
"`status` lists them as behind and `push --behind` sends it\n",
u.Module, shortCommit(u.Commit), id, readableList(on))
return nil
}
if decision.Together {
fmt.Printf("%s moved to %s; sending %s together\n",
u.Module, shortCommit(u.Commit), readableList(on))
return sendTo(ctx, f.open, on)
return askAgainOnGrants(sendTo(ctx, f.open, on))
}
// One at a time, and stopping at the first that fails.
//
@@ -69,13 +86,34 @@ func (f following) Upgraded(ctx context.Context, u link.Upgraded) error {
u.Module, shortCommit(u.Commit), readableList(on))
for _, node := range on {
if err := sendTo(ctx, f.open, []string{node}); err != nil {
return fmt.Errorf("%s did not take %s, so the machines after it were left alone: %w",
node, u.Module, err)
return askAgainOnGrants(fmt.Errorf("%s did not take %s, so the machines after it were left alone: %w",
node, u.Module, err))
}
}
return nil
}
// askAgainOnGrants marks a send that stopped at its grants as one to ask again (novox/hq issue 249):
// an announcement handled by a send whose memberships could not be issued is held and redelivered,
// rather than taken as handled with the machines left on the old version.
func askAgainOnGrants(err error) error {
if err != nil && errors.Is(err, errGrants) && !errors.Is(err, link.ErrTryAgain) {
return fmt.Errorf("%w: %w", link.ErrTryAgain, err)
}
return err
}
// rolledOutByAPlan is the open plan that will send a module's machines its new build — one holding
// the module that has not finished sending it — or empty when none will (novox/hq issue 249).
func rolledOutByAPlan(plans []inventory.Plan, module string) string {
for _, p := range plans {
if s, holds := p.Modules[module]; p.Open() && holds && (s == nil || (s.SentAt == nil && s.State != "failed")) {
return p.ID
}
}
return ""
}
// readableList names machines the way a sentence does, because this is read by a person deciding
// whether an upgrade went where they expected.
func readableList(names []string) string {
@@ -323,6 +361,45 @@ func (f following) SourceMoved(ctx context.Context, m link.SourceMoved) error {
}
defer release()
plan := planOfMerge(m, movedNames, edges)
// **A newer plan supersedes the older open plans of this repository and branch** (novox/hq issue
// 254, ADR 0218): what they had not built is planned here again, and they are closed, so one plan
// works a repository's modules at a time and a stuck one ends at the next merge.
working, err := inv.OpenPlans(ctx)
if err != nil {
return notNow(err)
}
rollsOut := func(module string) bool {
u, err := inv.UpgradeOf(ctx, module)
return err == nil && u.RollOut
}
folded, superseded := supersededBy(plan, working, rollsOut)
if len(folded) > 0 {
held := map[string]bool{}
for _, e := range entries {
held[e.Manifest.Module] = true
}
names := map[string]bool{}
for _, name := range movedNames {
names[name] = true
}
var also []string
for _, name := range folded {
// One the catalogue no longer holds would fail the newer plan's ask; it is not this
// merge's to build.
if held[name] && !names[name] {
names[name] = true
movedNames = append(movedNames, name)
also = append(also, name)
}
}
if len(also) > 0 {
again := planOfMerge(m, movedNames, edges)
again.ID, again.Created = plan.ID, plan.Created
plan = again
fmt.Printf(" %s, left unbuilt by an older plan of %s, are planned here again\n",
strings.Join(also, ", "), plan.Repository)
}
}
if hasCycle(plan.Tiers, edges) {
fmt.Printf(" the last tier depends on itself: %s — built together, in no order\n",
strings.Join(plan.Tiers[len(plan.Tiers)-1], ", "))
@@ -330,6 +407,14 @@ func (f following) SourceMoved(ctx context.Context, m link.SourceMoved) error {
if err := inv.SavePlan(ctx, plan); err != nil {
return notNow(err)
}
// Closed after the newer plan is kept, never before: a controller replaced between the two leaves
// both open, which the next merge settles, rather than neither.
for _, old := range superseded {
if err := inv.SavePlan(ctx, old); err != nil {
return notNow(err)
}
fmt.Printf(" %s (%s at %s) is %s\n", old.ID, old.Repository, short(old.Commit), old.Note)
}
var tiers []string
for i, t := range plan.Tiers {
tiers = append(tiers, fmt.Sprintf("%d: %s", i, strings.Join(t, ", ")))
@@ -423,25 +508,45 @@ func lastLookAt(entries []inventory.Entry, m link.SourceMoved) time.Time {
//
// A change inside *another* module's directory is that module's business and not this one's, even
// when the mesh does not hold that module: `known` is every module this repository is known to hold,
// whatever branch it was registered from. That is also the limit of this — a repository whose shared
// code sits inside a directory the mesh has never seen a module in reads as shared, and everything
// is rebuilt. Rebuilding too much is the safe direction: the fault this whole path exists for is a
// mesh that believes it is current and is not (novox/hq 04-ISSUES/131).
// whatever branch it was registered from.
//
// **And a module the mesh has never seen is still a module** (novox/hq issue 252). A merge adding a
// new module to the catalogue repository — `modules/newmod/module.json` and its files — read as a
// change to shared code, because `modules/newmod` was nobody's known directory, and every module
// built from the repository was rebuilt and rolled out for a module none of them is. So the
// directories that hold modules are known too: the parents of the known modules' directories
// (`modules`, never the root). A changed path `<parent>/<name>/…` belongs to the module at
// `<parent>/<name>` — held or not — and rebuilds nothing else, **provided it is shown to be a
// module**: its `module.json` is among the changed files (added, changed, or removed with it). A
// directory under the same parent whose manifest the merge did not touch may as well be a shared
// library (`modules/lib`), and that is still read as shared. Rebuilding too much remains the safe
// direction: the fault this whole path exists for is a mesh that believes it is current and is not
// (novox/hq 04-ISSUES/131). A file at the root, or directly in a parent, is shared as it always was.
func whatTheMergeTouched(candidates, known []inventory.Entry, m link.SourceMoved) []inventory.Entry {
// Nothing said about the files, or not all of them said: everything built from it is affected.
if len(m.Paths) == 0 || m.PathsTruncated {
return candidates
}
var dirs []string
parents := map[string]bool{}
for _, e := range known {
if e.Source.Path != "" && sameRepository(e.Source.Repository, m) {
dirs = append(dirs, e.Source.Path)
dir := strings.Trim(e.Source.Path, "/")
dirs = append(dirs, dir)
if parent := path.Dir(dir); parent != "." && parent != "/" {
parents[parent] = true
}
}
}
changed := map[string]bool{}
for _, p := range m.Paths {
if !insideAny(p, dirs) {
return candidates
changed[strings.TrimPrefix(p, "/")] = true
}
for _, p := range m.Paths {
if insideAny(p, dirs) || inAModuleOfItsOwn(p, parents, changed) {
continue
}
return candidates
}
var out []inventory.Entry
for _, e := range candidates {
@@ -452,6 +557,28 @@ func whatTheMergeTouched(candidates, known []inventory.Entry, m link.SourceMoved
return out
}
// inAModuleOfItsOwn is whether a changed file is inside a module directory the mesh does not know —
// `<parent>/<name>/…` under a directory known to hold modules, whose `module.json` the same merge
// changed (novox/hq issue 252). Such a file is that module's business and nobody else's.
func inAModuleOfItsOwn(p string, parents, changed map[string]bool) bool {
p = strings.TrimPrefix(p, "/")
for parent := range parents {
rest, under := strings.CutPrefix(p, parent+"/")
if !under {
continue
}
name, _, inADirectory := strings.Cut(rest, "/")
if !inADirectory || name == "" {
// A file directly in the parent — `modules/README.md` — is about all of them.
continue
}
if changed[parent+"/"+name+"/module.json"] {
return true
}
}
return false
}
// inside is whether a changed file is in a directory: that directory itself, or under it.
func inside(path, dir string) bool {
dir = strings.Trim(dir, "/")
+345
View File
@@ -0,0 +1,345 @@
package artifacts
import (
"bytes"
"context"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"fmt"
"io"
"net/http"
"strconv"
"strings"
"time"
"github.com/novox/mesh-controller/internal/catalogue"
)
// Every archive the store keeps is held by a manifest (novox/hq issue 253, ADR 0189).
//
// **The store's collector marks only from manifests.** The mesh's store is a stock registry, and
// its nightly `registry garbage-collect` walks every manifest in every repository, marks the blobs
// those manifests name, and deletes every blob it did not mark. An image is a manifest, so what
// the mesh keeps of an image survives. An archive was not: the builder put it in the store as a
// bare blob — upload, then `PUT ?digest=` — and nothing in the store names it. To the collector a
// bare blob is unreferenced, so the first real collection would have deleted every archive the
// mesh holds, kept or not, and every machine pinning a bundle would have found it gone. The
// collector runs `--dry-run` until this is true.
//
// **So each archive gets a holder**: the smallest OCI image manifest that names it — the empty
// config, one layer, nothing else — put in the archive's own repository, by digest, untagged. The
// collector marks it and so keeps the archive; the sweep lets go of an archive by deleting its
// holder first, which is what lets the bytes go at the next collection.
//
// **Nothing a machine reads changes.** The recorded reference stays
// `artifact-store://<module>/<artifact>/blobs/sha256:…`, and machines fetch the blob exactly as
// before. The holder is the store's bookkeeping, not a second way to reach anything.
//
// **Deterministic, so it never needs recording.** The holder is composed from the archive's digest
// and size alone, in a fixed field order with no timestamps or annotations, so the sweep can
// compute which manifest holds any archive from the reference it already has plus one HEAD for the
// size. No schema change, no second record that could disagree with the store.
const (
// mediaManifest is the type a holder is put and asked for as.
mediaManifest = "application/vnd.oci.image.manifest.v1+json"
// mediaEmpty is the OCI empty descriptor's type: a config that says nothing, for a manifest
// whose only purpose is to name its layer.
mediaEmpty = "application/vnd.oci.empty.v1+json"
// emptyDigest is the digest of `{}`, the empty config's content, fixed by the OCI spec.
emptyDigest = "sha256:44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a"
// mediaArchive is the layer type an archive is held as. Every archive the builder publishes is
// `pack`'s gzipped tar, so this is the true type and not a placeholder — and it is a constant,
// not read from anywhere, because the holder must be recomputable from the reference alone.
mediaArchive = "application/vnd.oci.image.layer.v1.tar+gzip"
)
// emptyConfig is the content emptyDigest names.
var emptyConfig = []byte("{}")
// manifestAccept is what a manifest is asked for as. A registry answers a manifest HEAD only in a
// type the caller named, and answers 404 to a bare one for a manifest it holds perfectly well
// (measured 2026-09-28; internal/builder/registry.go says how that was found).
var manifestAccept = []string{
mediaManifest,
"application/vnd.docker.distribution.manifest.v2+json",
}
type descriptor struct {
MediaType string `json:"mediaType"`
Digest string `json:"digest"`
Size int64 `json:"size"`
}
type holderManifest struct {
SchemaVersion int `json:"schemaVersion"`
MediaType string `json:"mediaType"`
Config descriptor `json:"config"`
Layers []descriptor `json:"layers"`
}
// Holder is the manifest that holds an archive in the store, and its digest.
//
// A pure function of the archive's digest and size: the same two in give the same bytes out,
// always, because `encoding/json` writes a struct's fields in their declared order and there is
// nothing here that varies by when or where it was composed.
func Holder(digest string, size int64) (body []byte, holder string) {
body, err := json.Marshal(holderManifest{
SchemaVersion: 2,
MediaType: mediaManifest,
Config: descriptor{MediaType: mediaEmpty, Digest: emptyDigest, Size: int64(len(emptyConfig))},
Layers: []descriptor{{MediaType: mediaArchive, Digest: digest, Size: size}},
})
if err != nil {
// Marshalling a struct of strings and integers cannot fail.
panic(err)
}
sum := sha256.Sum256(body)
return body, "sha256:" + hex.EncodeToString(sum[:])
}
// Hold makes sure the store holds this archive by a manifest, and says whether it had to write one.
//
// Takes a reference as the mesh records it. An image is its own manifest and needs no holder, so
// it answers false and nothing is asked. Idempotent: a holder already there is left alone, which
// is what lets the sweep run it over every kept archive on every build and so backfill the bare
// blobs published before holders existed (novox/hq issue 253).
//
// Gone when the store does not hold the archive at all: there is nothing to hold, and that is a
// fact the caller reports rather than one this invents a remedy for.
func (s Store) Hold(ctx context.Context, reference string) (bool, error) {
repository, digest, archive, err := s.archive(reference)
if err != nil || !archive {
return false, err
}
size, err := s.blobSize(ctx, repository, digest)
if err != nil {
return false, err
}
return s.HoldBlob(ctx, repository, digest, size)
}
// Held is whether the store holds this archive by its manifest. Asks and changes nothing — the
// question an operator needs answered with "none unheld" before the collector is let loose.
//
// An image answers true: it is its own manifest. An archive the store does not have answers Gone.
func (s Store) Held(ctx context.Context, reference string) (bool, error) {
repository, digest, archive, err := s.archive(reference)
if err != nil {
return false, err
}
if !archive {
return true, nil
}
size, err := s.blobSize(ctx, repository, digest)
if err != nil {
return false, err
}
_, holder := Holder(digest, size)
return s.has(ctx, s.url(repository, "manifests", holder), manifestAccept...)
}
// HoldBlob puts the holder for a blob of this digest and size into its repository, unless it is
// there already. Answers whether it wrote one.
//
// The builder calls this with the size it has just uploaded; the sweep, through Hold, with the size
// the store reports. Both arrive at the same holder, which is the point of composing it.
func (s Store) HoldBlob(ctx context.Context, repository, digest string, size int64) (bool, error) {
if s.Address == "" {
return false, fmt.Errorf("this mesh has no artifact store on its network to hold %s/%s in", repository, digest)
}
body, holder := Holder(digest, size)
there, err := s.has(ctx, s.url(repository, "manifests", holder), manifestAccept...)
if err != nil {
return false, err
}
if there {
return false, nil
}
// The config must be in the repository before a manifest naming it is accepted: a registry
// refuses a manifest whose blobs it cannot find there, which is the property that makes a
// holder mean something.
if err := s.putBlob(ctx, repository, emptyDigest, emptyConfig); err != nil {
return false, err
}
// **By digest, never by tag.** A tag would be one more name to move and one more thing the
// collector's `--delete-untagged` would read as meaningful; the mesh names nothing by tag that
// it pins by digest, and an untagged manifest is kept by plain collection.
request, err := http.NewRequestWithContext(ctx, http.MethodPut,
s.url(repository, "manifests", holder), bytes.NewReader(body))
if err != nil {
return false, err
}
request.Header.Set("Content-Type", mediaManifest)
response, err := s.client().Do(request)
if err != nil {
return false, err
}
defer response.Body.Close()
if response.StatusCode != http.StatusCreated {
said, _ := io.ReadAll(io.LimitReader(response.Body, 4096))
return false, fmt.Errorf("the artifact store refused to hold %s/%s: %s %s",
repository, digest, response.Status, strings.TrimSpace(string(said)))
}
return true, nil
}
// letGoOfHolder deletes the manifest holding an archive, before the archive's own link goes.
//
// **Holder first.** Deleting the blob link alone leaves a manifest still naming the blob, and the
// collector would keep its bytes for ever on the strength of it — the sweep would record the
// archive collected while the disk said otherwise. Deleting the holder first and failing before
// the link goes leaves an unheld archive that the next sweep still offers, which is safe.
//
// A store that no longer has the blob answers Gone: without its size the holder cannot be named,
// and without the blob there is nothing left for a holder to keep. A store that never had a holder
// for it — an archive published before holders, never backfilled — answers 404 to the delete, and
// that is the outcome wanted.
func (s Store) letGoOfHolder(ctx context.Context, repository, digest string) error {
size, err := s.blobSize(ctx, repository, digest)
if err != nil {
return err
}
_, holder := Holder(digest, size)
err = s.remove(ctx, s.url(repository, "manifests", holder), repository+"/manifests/"+holder)
if err == Gone {
return nil
}
return err
}
// archive reads a recorded reference into its repository and digest, and whether it is an archive
// at all. Refuses as ErrNotOurs anything the mesh did not put in its own store.
func (s Store) archive(reference string) (repository, digest string, archive bool, err error) {
path, kept := catalogue.InArtifactStore(reference)
if !kept {
return "", "", false, fmt.Errorf("%w: %s", ErrNotOurs, reference)
}
if s.Address == "" {
return "", "", false, fmt.Errorf("this mesh has no artifact store on its network to ask about %s", reference)
}
repository, kind, digest, err := split(path)
if err != nil {
return "", "", false, err
}
return repository, digest, kind == "blobs", nil
}
// blobSize is how large the store says a blob is; Gone when it does not have it.
func (s Store) blobSize(ctx context.Context, repository, digest string) (int64, error) {
request, err := http.NewRequestWithContext(ctx, http.MethodHead, s.url(repository, "blobs", digest), nil)
if err != nil {
return 0, err
}
response, err := s.client().Do(request)
if err != nil {
return 0, fmt.Errorf("cannot reach the artifact store at %s: %w", s.Address, err)
}
defer response.Body.Close()
switch response.StatusCode {
case http.StatusOK:
case http.StatusNotFound:
return 0, Gone
default:
return 0, fmt.Errorf("the artifact store answered %s for %s/blobs/%s", response.Status, repository, digest)
}
// Read from the header rather than ContentLength: a HEAD's ContentLength is what the response
// says it would have sent, which Go reports faithfully, but a proxy in between is free to drop
// it, and the header is what the registry itself wrote.
if length := response.Header.Get("Content-Length"); length != "" {
if n, err := strconv.ParseInt(length, 10, 64); err == nil && n >= 0 {
return n, nil
}
}
if response.ContentLength >= 0 {
return response.ContentLength, nil
}
return 0, fmt.Errorf("the artifact store holds %s/blobs/%s and will not say how large it is", repository, digest)
}
// putBlob uploads a small blob unless the repository already has it: ask where, then put it there
// naming the digest — the registry's own two steps, the same the builder takes for an archive.
func (s Store) putBlob(ctx context.Context, repository, digest string, body []byte) error {
if there, err := s.has(ctx, s.url(repository, "blobs", digest)); err != nil {
return err
} else if there {
return nil
}
start, err := http.NewRequestWithContext(ctx, http.MethodPost,
"http://"+s.Address+"/v2/"+repository+"/blobs/uploads/", nil)
if err != nil {
return err
}
begun, err := s.client().Do(start)
if err != nil {
return fmt.Errorf("cannot start an upload to %s: %w", repository, err)
}
begun.Body.Close()
if begun.StatusCode != http.StatusAccepted {
return fmt.Errorf("the artifact store answered %s when asked where to put a blob in %s", begun.Status, repository)
}
where := begun.Header.Get("Location")
if where == "" {
return fmt.Errorf("the artifact store accepted an upload to %s and said nowhere to put it", repository)
}
if strings.HasPrefix(where, "/") {
where = "http://" + s.Address + where
}
separator := "?"
if strings.Contains(where, "?") {
separator = "&"
}
put, err := http.NewRequestWithContext(ctx, http.MethodPut, where+separator+"digest="+digest, bytes.NewReader(body))
if err != nil {
return err
}
put.Header.Set("Content-Type", "application/octet-stream")
done, err := s.client().Do(put)
if err != nil {
return err
}
defer done.Body.Close()
if done.StatusCode != http.StatusCreated {
said, _ := io.ReadAll(io.LimitReader(done.Body, 4096))
return fmt.Errorf("the artifact store refused a blob in %s: %s %s", repository, done.Status, strings.TrimSpace(string(said)))
}
return nil
}
// has is whether the store answers 200 for a HEAD at that URL.
func (s Store) has(ctx context.Context, url string, accept ...string) (bool, error) {
request, err := http.NewRequestWithContext(ctx, http.MethodHead, url, nil)
if err != nil {
return false, err
}
for _, media := range accept {
request.Header.Add("Accept", media)
}
response, err := s.client().Do(request)
if err != nil {
return false, fmt.Errorf("cannot reach the artifact store at %s: %w", s.Address, err)
}
defer response.Body.Close()
switch response.StatusCode {
case http.StatusOK:
return true, nil
case http.StatusNotFound:
return false, nil
default:
return false, fmt.Errorf("the artifact store answered %s for %s", response.Status, url)
}
}
func (s Store) url(repository, kind, digest string) string {
return "http://" + s.Address + "/v2/" + repository + "/" + kind + "/" + digest
}
func (s Store) client() *http.Client {
if s.HTTP != nil {
return s.HTTP
}
return &http.Client{Timeout: 30 * time.Second}
}
+268
View File
@@ -0,0 +1,268 @@
package artifacts
import (
"bytes"
"context"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"io"
"net/http"
"net/http/httptest"
"strconv"
"strings"
"sync"
"testing"
"github.com/novox/mesh-controller/internal/catalogue"
)
// Every kept archive is held by a manifest (novox/hq issue 253, ADR 0189).
//
// Against an in-memory registry that keeps blobs and manifests per repository and refuses what a
// registry refuses — a blob whose digest does not match, a manifest whose digest does not match
// or whose blobs the repository does not have, a manifest asked for without an Accept naming its
// type. What is asserted is this side's decisions; the live test below asserts the registry's.
type memRegistry struct {
mu sync.Mutex
blobs map[string][]byte // repository + "@" + digest
manifests map[string][]byte // repository + "@" + digest
writes []string // every PUT and DELETE, as "METHOD path"
}
func digestOf(body []byte) string {
sum := sha256.Sum256(body)
return "sha256:" + hex.EncodeToString(sum[:])
}
func (m *memRegistry) serve(t *testing.T) Store {
t.Helper()
m.blobs = map[string][]byte{}
m.manifests = map[string][]byte{}
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
m.mu.Lock()
defer m.mu.Unlock()
path := strings.TrimPrefix(r.URL.Path, "/v2/")
if r.Method == http.MethodPut || r.Method == http.MethodDelete {
m.writes = append(m.writes, r.Method+" "+r.URL.Path)
}
switch {
case r.Method == http.MethodPost && strings.HasSuffix(path, "/blobs/uploads/"):
repository := strings.TrimSuffix(path, "/blobs/uploads/")
w.Header().Set("Location", "/upload/"+repository+"?state=x")
w.WriteHeader(http.StatusAccepted)
case r.Method == http.MethodPut && strings.HasPrefix(r.URL.Path, "/upload/"):
repository := strings.TrimPrefix(r.URL.Path, "/upload/")
body, _ := io.ReadAll(r.Body)
digest := r.URL.Query().Get("digest")
if digest != digestOf(body) {
w.WriteHeader(http.StatusBadRequest)
return
}
m.blobs[repository+"@"+digest] = body
w.WriteHeader(http.StatusCreated)
case strings.Contains(path, "/blobs/"):
repository, digest, _ := strings.Cut(path, "/blobs/")
key := repository + "@" + digest
body, ok := m.blobs[key]
if !ok {
w.WriteHeader(http.StatusNotFound)
return
}
switch r.Method {
case http.MethodHead:
w.Header().Set("Content-Length", strconv.Itoa(len(body)))
w.WriteHeader(http.StatusOK)
case http.MethodDelete:
delete(m.blobs, key)
w.WriteHeader(http.StatusAccepted)
default:
w.WriteHeader(http.StatusMethodNotAllowed)
}
case strings.Contains(path, "/manifests/"):
repository, digest, _ := strings.Cut(path, "/manifests/")
key := repository + "@" + digest
switch r.Method {
case http.MethodHead:
if _, ok := m.manifests[key]; !ok || !strings.Contains(r.Header.Get("Accept"), mediaManifest) {
w.WriteHeader(http.StatusNotFound)
return
}
w.WriteHeader(http.StatusOK)
case http.MethodPut:
body, _ := io.ReadAll(r.Body)
if digest != digestOf(body) {
w.WriteHeader(http.StatusBadRequest)
return
}
var named holderManifest
if err := json.Unmarshal(body, &named); err != nil {
w.WriteHeader(http.StatusBadRequest)
return
}
for _, d := range append([]descriptor{named.Config}, named.Layers...) {
if _, ok := m.blobs[repository+"@"+d.Digest]; !ok {
w.WriteHeader(http.StatusBadRequest)
fmt.Fprintf(w, "MANIFEST_BLOB_UNKNOWN %s", d.Digest)
return
}
}
m.manifests[key] = body
w.WriteHeader(http.StatusCreated)
case http.MethodDelete:
if _, ok := m.manifests[key]; !ok {
w.WriteHeader(http.StatusNotFound)
return
}
delete(m.manifests, key)
w.WriteHeader(http.StatusAccepted)
}
default:
w.WriteHeader(http.StatusNotFound)
}
}))
t.Cleanup(server.Close)
return Store{Address: strings.TrimPrefix(server.URL, "http://")}
}
// bare puts an archive in the store the way the builder did before holders: a blob, nothing more.
func (m *memRegistry) bare(repository string, body []byte) string {
m.mu.Lock()
defer m.mu.Unlock()
digest := digestOf(body)
m.blobs[repository+"@"+digest] = body
return catalogue.ArtifactStoreScheme + repository + "/blobs/" + digest
}
func TestTheHolderIsComposedFromTheDigestAndSizeAlone(t *testing.T) {
// The sweep must arrive at the very manifest the builder wrote, with nothing recorded between
// them. Same inputs, same bytes — and a different size is a different holder, so a holder can
// never be mistaken for one of a different blob.
digest := "sha256:" + strings.Repeat("a", 64)
one, first := Holder(digest, 42)
two, second := Holder(digest, 42)
if !bytes.Equal(one, two) || first != second {
t.Fatalf("the same archive composed two holders:\n%s\n%s", one, two)
}
if _, other := Holder(digest, 43); other == first {
t.Fatal("a different size composed the same holder")
}
want := `{"schemaVersion":2,"mediaType":"application/vnd.oci.image.manifest.v1+json",` +
`"config":{"mediaType":"application/vnd.oci.empty.v1+json",` +
`"digest":"sha256:44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a","size":2},` +
`"layers":[{"mediaType":"application/vnd.oci.image.layer.v1.tar+gzip","digest":"` + digest + `","size":42}]}`
if string(one) != want {
t.Fatalf("the holder is\n%s\nwant\n%s", one, want)
}
if digestOf(emptyConfig) != emptyDigest {
t.Fatalf("the empty config's digest is %s, not %s", digestOf(emptyConfig), emptyDigest)
}
}
func TestHoldBackfillsABareArchiveAndIsIdempotent(t *testing.T) {
// The archives published before this have no holder. Hold, run over every kept archive on
// every sweep, writes one the first time and nothing after.
m := &memRegistry{}
store := m.serve(t)
ctx := context.Background()
body := []byte("a theme")
reference := m.bare("shell/config", body)
if held, err := store.Held(ctx, reference); err != nil || held {
t.Fatalf("a bare blob reads as held=%v (%v)", held, err)
}
wrote, err := store.Hold(ctx, reference)
if err != nil {
t.Fatal(err)
}
if !wrote {
t.Fatal("holding a bare archive wrote nothing")
}
_, holder := Holder(digestOf(body), int64(len(body)))
if _, ok := m.manifests["shell/config@"+holder]; !ok {
t.Fatalf("the store holds manifests %v; want %s", m.manifests, holder)
}
if held, err := store.Held(ctx, reference); err != nil || !held {
t.Fatalf("after holding, held=%v (%v)", held, err)
}
writes := len(m.writes)
wrote, err = store.Hold(ctx, reference)
if err != nil {
t.Fatal(err)
}
if wrote || len(m.writes) != writes {
t.Fatalf("holding again wrote %v", m.writes[writes:])
}
}
func TestAnImageNeedsNoHolderAndAMissingArchiveIsGone(t *testing.T) {
m := &memRegistry{}
store := m.serve(t)
ctx := context.Background()
// An image is its own manifest: nothing is asked.
wrote, err := store.Hold(ctx, catalogue.ArtifactStoreScheme+"web/app@sha256:"+strings.Repeat("b", 64))
if err != nil || wrote || len(m.writes) != 0 {
t.Fatalf("holding an image wrote=%v err=%v writes=%v", wrote, err, m.writes)
}
// An archive the store does not have is a fact to report, not something to invent a holder for.
_, err = store.Hold(ctx, catalogue.ArtifactStoreScheme+"web/config/blobs/sha256:"+strings.Repeat("c", 64))
if !errors.Is(err, Gone) {
t.Fatalf("holding a missing archive answered %v, want Gone", err)
}
// And a reference that is not the mesh's is refused as such.
if _, err := store.Hold(ctx, "docker.io/library/registry@sha256:abc"); !errors.Is(err, ErrNotOurs) {
t.Fatalf("holding a vendor's image answered %v, want ErrNotOurs", err)
}
}
func TestLettingGoOfAnArchiveDeletesItsHolderFirst(t *testing.T) {
// A holder left behind would keep the bytes through every collection while the record said
// collected; the link deleted first and the holder failing after would be that exactly.
m := &memRegistry{}
store := m.serve(t)
ctx := context.Background()
body := []byte("an old theme")
reference := m.bare("shell/config", body)
if _, err := store.Hold(ctx, reference); err != nil {
t.Fatal(err)
}
m.writes = nil
if err := store.LetGo(ctx, reference); err != nil {
t.Fatal(err)
}
_, holder := Holder(digestOf(body), int64(len(body)))
want := []string{
"DELETE /v2/shell/config/manifests/" + holder,
"DELETE /v2/shell/config/blobs/" + digestOf(body),
}
if strings.Join(m.writes, "\n") != strings.Join(want, "\n") {
t.Fatalf("the store was asked\n%s\nwant\n%s", strings.Join(m.writes, "\n"), strings.Join(want, "\n"))
}
if len(m.manifests) != 0 {
t.Fatalf("a holder survived: %v", m.manifests)
}
// Asked again, the archive is already gone, which is the outcome wanted.
if err := store.LetGo(ctx, reference); !errors.Is(err, Gone) {
t.Fatalf("letting go twice answered %v, want Gone", err)
}
}
func TestLettingGoOfAnUnheldArchiveStillDeletesIt(t *testing.T) {
// An archive published before holders and let go of before any sweep held it: the holder's
// delete answers 404, which is the outcome wanted, and the blob still goes.
m := &memRegistry{}
store := m.serve(t)
reference := m.bare("shell/config", []byte("never held"))
if err := store.LetGo(context.Background(), reference); err != nil {
t.Fatal(err)
}
if len(m.blobs) != 0 {
t.Fatalf("the blob survived: %v", m.blobs)
}
}
+130
View File
@@ -0,0 +1,130 @@
package artifacts
import (
"bytes"
"context"
"fmt"
"io"
"net/http"
"os"
"os/exec"
"strings"
"testing"
"time"
"github.com/novox/mesh-controller/internal/catalogue"
)
// The registry's own collector keeps a held archive and takes a bare one (novox/hq issue 253,
// ADR 0189).
//
// Everything else here is asserted against a fake, which can only say what this side asks. This is
// the one question a fake cannot answer — what `registry garbage-collect` actually does with what
// this side wrote — and it is the whole of whether the store's nightly step may stop being a dry
// run. Against the very image the mesh's store runs:
//
// docker run -d --rm --name mesh-controller-registry -p 15000:5000 \
// -e REGISTRY_STORAGE_DELETE_ENABLED=true registry:2.8.3
// MESH_TEST_REGISTRY=127.0.0.1:15000 MESH_TEST_REGISTRY_CONTAINER=mesh-controller-registry \
// go test -run Live ./internal/artifacts/
// docker stop mesh-controller-registry
//
// Skipped without both variables: it needs a registry it may write to and collect, and a container
// to run the collector in.
func TestLiveTheRegistrysCollectorKeepsWhatIsHeldAndTakesWhatIsNot(t *testing.T) {
address := os.Getenv("MESH_TEST_REGISTRY")
container := os.Getenv("MESH_TEST_REGISTRY_CONTAINER")
if address == "" || container == "" {
t.Skip("no MESH_TEST_REGISTRY / MESH_TEST_REGISTRY_CONTAINER; see this test's comment for the registry to raise")
}
ctx := context.Background()
store := Store{Address: address}
run := time.Now().UnixNano()
// Four archives in four repositories, each a different story. Distinct bytes per run, so a
// registry reused across runs cannot answer for an earlier one.
put := func(name string) (repository, digest string, body []byte) {
repository = fmt.Sprintf("live-%d/%s", run, name)
body = []byte(fmt.Sprintf("%s archive of run %d", name, run))
digest = digestOf(body)
if err := store.putBlob(ctx, repository, digest, body); err != nil {
t.Fatal(err)
}
return repository, digest, body
}
reference := func(repository, digest string) string {
return catalogue.ArtifactStoreScheme + repository + "/blobs/" + digest
}
// Published held — what PublishArchive now does.
heldRepo, heldDigest, heldBody := put("held")
if _, err := store.HoldBlob(ctx, heldRepo, heldDigest, int64(len(heldBody))); err != nil {
t.Fatalf("the registry refused a holder: %v", err)
}
// Published bare, as before, and never held: what the collector must take.
_, bareDigest, _ := put("bare")
// Published bare and then held by the sweep: the backfill.
backRepo, backDigest, backBody := put("backfilled")
if wrote, err := store.Hold(ctx, reference(backRepo, backDigest)); err != nil || !wrote {
t.Fatalf("backfilling wrote=%v: %v", wrote, err)
}
if held, err := store.Held(ctx, reference(backRepo, backDigest)); err != nil || !held {
t.Fatalf("after backfilling, held=%v: %v", held, err)
}
// Held, and then let go of by the sweep: holder first, then the link.
goneRepo, goneDigest, _ := put("let-go")
if _, err := store.Hold(ctx, reference(goneRepo, goneDigest)); err != nil {
t.Fatal(err)
}
if err := store.LetGo(ctx, reference(goneRepo, goneDigest)); err != nil {
t.Fatalf("letting go of a held archive: %v", err)
}
collected, err := exec.CommandContext(ctx, "docker", "exec", container,
"registry", "garbage-collect", "/etc/docker/registry/config.yml").CombinedOutput()
if err != nil {
t.Fatalf("the collector failed: %v\n%s", err, collected)
}
t.Logf("the collector said:\n%s", lastLines(string(collected), 12))
// What is asserted is the bytes on the store's disk, not what the running server answers: the
// server caches blob descriptors in memory and can answer for a blob the collector removed.
onDisk := func(digest string) bool {
hex := strings.TrimPrefix(digest, "sha256:")
path := "/var/lib/registry/docker/registry/v2/blobs/sha256/" + hex[:2] + "/" + hex + "/data"
return exec.CommandContext(ctx, "docker", "exec", container, "test", "-f", path).Run() == nil
}
if !onDisk(heldDigest) {
t.Error("the collector took an archive published held")
}
if !onDisk(backDigest) {
t.Error("the collector took an archive the sweep backfilled a holder for")
}
if onDisk(bareDigest) {
t.Error("the collector kept a bare archive — then the holders prove nothing, and this test is wrong")
}
if onDisk(goneDigest) {
t.Error("the collector kept an archive the sweep let go of: its holder outlived its link")
}
// And what survived is still fetched exactly as machines fetch it: the blob, by digest.
for repository, want := range map[string][]byte{heldRepo: heldBody, backRepo: backBody} {
digest := digestOf(want)
response, err := http.Get(catalogue.Routed(reference(repository, digest), address))
if err != nil {
t.Fatal(err)
}
got, _ := io.ReadAll(response.Body)
response.Body.Close()
if response.StatusCode != http.StatusOK || !bytes.Equal(got, want) {
t.Errorf("%s answered %s with %q after collection", repository, response.Status, got)
}
}
}
func lastLines(s string, n int) string {
lines := strings.Split(strings.TrimSpace(s), "\n")
if len(lines) > n {
lines = lines[len(lines)-n:]
}
return strings.Join(lines, "\n")
}
+20 -10
View File
@@ -1,7 +1,8 @@
// Package artifacts speaks to the mesh's artifact store over its own door.
//
// Only what the mesh needs that nothing else does: letting go of something it put there
// (novox/hq ADR 0189, issue 108). Pushing is the builder's, through the container runtime; reading
// (novox/hq ADR 0189, issue 108), and holding every archive it keeps by a manifest so the store's
// own collector does not take it (novox/hq issue 253). Pushing is the builder's, through the container runtime; reading
// is every machine's, through its runtime. This is the one operation that belongs to the thing
// holding the records, because it is the only one that is a decision rather than a transfer.
package artifacts
@@ -12,7 +13,6 @@ import (
"fmt"
"net/http"
"strings"
"time"
"github.com/novox/mesh-controller/internal/catalogue"
)
@@ -43,6 +43,9 @@ var ErrNotOurs = errors.New("not a reference into the mesh's artifact store")
// an image, `…/blobs/sha256:…` for an archive — because that is the identity every record uses,
// and composes the address here at the moment of use.
//
// An archive is let go of in two deletes, its holder manifest and then the blob's link (novox/hq
// issue 253); an image in one.
//
// Returns Gone when the store answers that it does not have it. That is not a failure: the sweep
// wants the artifact absent, and it is. It is distinguished from success only so a caller can say
// which of the two happened.
@@ -66,17 +69,24 @@ func (s Store) LetGo(ctx context.Context, reference string) error {
if err != nil {
return err
}
url := "http://" + s.Address + "/v2/" + repository + "/" + kind + "/" + digest
if kind == "blobs" {
// **An archive's holder goes before the archive** (novox/hq issue 253): a manifest left
// naming the blob would keep its bytes through every collection while the record said
// collected. Gone here means the store has no such blob, so there is nothing to let go.
if err := s.letGoOfHolder(ctx, repository, digest); err != nil {
return err
}
}
return s.remove(ctx, s.url(repository, kind, digest), reference)
}
// remove asks the store to delete what is at url. Gone when it has no such thing.
func (s Store) remove(ctx context.Context, url, what string) error {
request, err := http.NewRequestWithContext(ctx, http.MethodDelete, url, nil)
if err != nil {
return err
}
client := s.HTTP
if client == nil {
client = &http.Client{Timeout: 30 * time.Second}
}
response, err := client.Do(request)
response, err := s.client().Do(request)
if err != nil {
return err
}
@@ -92,9 +102,9 @@ func (s Store) LetGo(ctx context.Context, reference string) error {
return fmt.Errorf(
"the artifact store refuses deletion: its server was started without it enabled "+
"(REGISTRY_STORAGE_DELETE_ENABLED), so nothing can be collected until the store "+
"module is applied again (novox/hq ADR 0189). Asking about %s", reference)
"module is applied again (novox/hq ADR 0189). Asking about %s", what)
default:
return fmt.Errorf("the artifact store answered %s for %s", response.Status, reference)
return fmt.Errorf("the artifact store answered %s for %s", response.Status, what)
}
}
+19 -2
View File
@@ -20,6 +20,17 @@ func fakeStore(t *testing.T, answer int) (Store, *[]string) {
t.Helper()
var asked []string
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodHead && strings.Contains(r.URL.Path, "/blobs/") {
// An archive's size, asked so its holder can be named (novox/hq issue 253). A store
// that does not have the thing does not have its blob either.
if answer == http.StatusNotFound {
w.WriteHeader(http.StatusNotFound)
return
}
w.Header().Set("Content-Length", "7")
w.WriteHeader(http.StatusOK)
return
}
if r.Method != http.MethodDelete {
t.Errorf("the store was asked %s %s; collecting is a delete", r.Method, r.URL.Path)
}
@@ -45,8 +56,14 @@ func TestAnImageAndAnArchiveAreAskedForAtTheirOwnEndpoints(t *testing.T) {
if err := store.LetGo(ctx, archive); err != nil {
t.Fatal(err)
}
want := []string{"/v2/web/app/manifests/sha256:abc123", "/v2/web/config/blobs/sha256:def456"}
if len(*asked) != 2 || (*asked)[0] != want[0] || (*asked)[1] != want[1] {
// The archive's holder goes first, then the archive (novox/hq issue 253).
_, holder := Holder("sha256:def456", 7)
want := []string{
"/v2/web/app/manifests/sha256:abc123",
"/v2/web/config/manifests/" + holder,
"/v2/web/config/blobs/sha256:def456",
}
if strings.Join(*asked, " ") != strings.Join(want, " ") {
t.Fatalf("the store was asked %v; want %v", *asked, want)
}
}
+91
View File
@@ -0,0 +1,91 @@
package broker
import (
"fmt"
"os"
"testing"
"time"
"github.com/nats-io/nats.go"
)
// A consumer that reacts to announcements, made on a stream that keeps a week of them, starts from now:
// made from the start it replays every merge and build of that week (novox/hq issue 248). And a reset
// re-makes a stuck one from now, its configuration otherwise kept, and refuses a work queue.
//
// docker run -d --rm --name t -p 14231:4222 nats:2.10-alpine -js
// MESH_TEST_NATS=nats://127.0.0.1:14231 go test ./internal/broker/ -run TestAConsumerMadeFromNow
func TestAConsumerMadeFromNowNeverReplaysTheStreamsHistory(t *testing.T) {
url := os.Getenv("MESH_TEST_NATS")
if url == "" {
t.Skip("MESH_TEST_NATS unset")
}
js, err := Dial(url)
if err != nil {
t.Fatal(err)
}
defer js.Close()
const stream = "HISTORY_TEST"
_ = js.js.DeleteStream(stream)
if _, err := js.js.AddStream(&nats.StreamConfig{Name: stream, Subjects: []string{"history.>"}, Storage: nats.MemoryStorage}); err != nil {
t.Fatal(err)
}
defer func() { _ = js.js.DeleteStream(stream) }()
for i := 0; i < 5; i++ {
if _, err := js.js.Publish("history.merged", []byte(fmt.Sprint(i))); err != nil {
t.Fatal(err)
}
}
fresh := Consumer{Name: "fresh", Stream: stream, Filters: []string{"history.merged"}, Push: true,
AckWaitSeconds: 30, MaxDeliver: 5, MaxAckPending: 1, FromNow: true}
if err := js.EnsureConsumer(fresh); err != nil {
t.Fatal(err)
}
info, _ := js.js.ConsumerInfo(stream, "fresh")
if info.NumPending != 0 || info.Config.DeliverPolicy != nats.DeliverNewPolicy {
t.Fatalf("a consumer made from now holds %d of the stream's past (policy %v)", info.NumPending, info.Config.DeliverPolicy)
}
_, _ = js.js.Publish("history.merged", []byte("new"))
time.Sleep(100 * time.Millisecond)
if info, _ = js.js.ConsumerInfo(stream, "fresh"); info.NumPending != 1 {
t.Fatalf("a new announcement is not pending: %d", info.NumPending)
}
// Asserted again, it keeps where it is.
if err := js.EnsureConsumer(fresh); err != nil {
t.Fatal(err)
}
// One made from the start, as the server's default makes it, and stuck behind its history.
old := fresh
old.Name, old.FromNow = "old", false
if err := js.EnsureConsumer(old); err != nil {
t.Fatal(err)
}
if info, _ = js.js.ConsumerInfo(stream, "old"); info.NumPending != 6 {
t.Fatalf("the default should replay all six: %d", info.NumPending)
}
before, after, err := js.ResetConsumer(stream, "old")
if err != nil {
t.Fatal(err)
}
info, _ = js.js.ConsumerInfo(stream, "old")
if before.Pending != 6 || after.Pending != 0 || info.Config.MaxAckPending != 1 || info.Config.MaxDeliver != 5 ||
info.Config.AckWait != 30*time.Second || info.Config.DeliverSubject == "" {
t.Fatalf("before %+v after %+v config %+v", before, after, info.Config)
}
// Never a work queue: what is pending there is work.
const queue = "QUEUE_TEST"
_ = js.js.DeleteStream(queue)
if _, err := js.js.AddStream(&nats.StreamConfig{Name: queue, Subjects: []string{"queue.>"}, Retention: nats.WorkQueuePolicy, Storage: nats.MemoryStorage}); err != nil {
t.Fatal(err)
}
defer func() { _ = js.js.DeleteStream(queue) }()
if _, err := js.js.AddConsumer(queue, &nats.ConsumerConfig{Durable: "w", AckPolicy: nats.AckExplicitPolicy}); err != nil {
t.Fatal(err)
}
if _, _, err := js.ResetConsumer(queue, "w"); err == nil {
t.Fatal("a work queue's consumer was reset")
}
}
+7 -1
View File
@@ -47,7 +47,13 @@ type Consumer struct {
// and a merge that came back rebuilt what it had just built, five times over on 2026-09-30.
// With one outstanding, the server holds the rest, and the heartbeat is keeping the message.
MaxAckPending int
Why string
// FromNow makes a consumer that does not exist yet start at the stream's end rather than its
// beginning (novox/hq issue 248). For a consumer that reacts to announcements — a merge, a build's
// outcome — on a stream that keeps a week of them: the server's default, everything the stream
// holds, replays every merge and every build of that week as if it had just happened. A consumer
// that exists keeps where it is, whatever this says; only its making is decided here.
FromNow bool
Why string
}
// seatStreamName is the stream holding a seat's inbound work. Named after the seat rather than
+48
View File
@@ -308,6 +308,9 @@ func (j *JetStream) EnsureConsumer(c Consumer) error {
}
return nil
case errors.Is(err, nats.ErrConsumerNotFound):
if c.FromNow {
want.DeliverPolicy = nats.DeliverNewPolicy
}
if _, err := j.js.AddConsumer(c.Stream, want); err != nil {
return fmt.Errorf("creating consumer %s on %s: %w", c.Name, c.Stream, err)
}
@@ -317,6 +320,51 @@ func (j *JetStream) EnsureConsumer(c Consumer) error {
}
}
// ConsumerState is what a reset says about a consumer, before and after.
type ConsumerState struct {
DeliverPolicy string
Delivered uint64
AckFloor uint64
Pending uint64
AckPending int
}
func stateOf(info *nats.ConsumerInfo) ConsumerState {
policy, _ := info.Config.DeliverPolicy.MarshalJSON()
return ConsumerState{DeliverPolicy: strings.Trim(string(policy), `"`), Delivered: info.Delivered.Stream,
AckFloor: info.AckFloor.Stream, Pending: info.NumPending, AckPending: info.NumAckPending}
}
// ResetConsumer re-makes a consumer to start from now, its configuration otherwise unchanged (novox/hq
// issue 248): what it had not yet delivered or acknowledged is dropped, which is the point — on a stream
// that keeps history, a consumer replaying a week of announcements does nothing anyone wants. Refused on
// a work queue, where what is pending is work nobody else will do.
func (j *JetStream) ResetConsumer(stream, name string) (before, after ConsumerState, err error) {
info, err := j.js.StreamInfo(stream)
if err != nil {
return before, after, fmt.Errorf("asking about stream %s: %w", stream, err)
}
if info.Config.Retention == nats.WorkQueuePolicy {
return before, after, fmt.Errorf("%s is a work queue: what its consumer has pending is work, and a reset would drop it", stream)
}
have, err := j.js.ConsumerInfo(stream, name)
if err != nil {
return before, after, fmt.Errorf("asking about consumer %s on %s: %w", name, stream, err)
}
before = stateOf(have)
want := have.Config
want.DeliverPolicy = nats.DeliverNewPolicy
want.OptStartSeq, want.OptStartTime = 0, nil
if err := j.js.DeleteConsumer(stream, name); err != nil {
return before, after, fmt.Errorf("removing consumer %s on %s: %w", name, stream, err)
}
made, err := j.js.AddConsumer(stream, &want)
if err != nil {
return before, after, fmt.Errorf("re-making consumer %s on %s — it is gone until the controller asserts it at its next start: %w", name, stream, err)
}
return before, stateOf(made), nil
}
func retentionOf(r Retention) nats.RetentionPolicy {
switch r {
case RetentionWorkQueue:
+1
View File
@@ -270,6 +270,7 @@ func MeshConsumers() []Consumer {
// announcement handed over behind it must wait on the server, not time out on the
// client and come back to be acted on again.
MaxAckPending: 1,
FromNow: true,
Why: "the two events the mesh's own controller reacts to, one at a time; after " +
"max-deliver it dead-letters, because an announcement it cannot act on will not " +
"become actionable",
+29 -7
View File
@@ -7,6 +7,8 @@ import (
"io"
"net/http"
"strings"
"github.com/novox/mesh-controller/internal/artifacts"
)
// Where built artifacts go.
@@ -66,22 +68,32 @@ func (r Registry) PublishImage(ctx context.Context, localTag, repository string)
return pinned, nil
}
// PublishArchive stores bytes as a blob and returns where to fetch them from.
// PublishArchive stores bytes as a blob, holds it by a manifest, and returns where to fetch them
// from.
//
// Two steps, which is the registry's own protocol: ask for somewhere to put it, then put it there
// naming the digest. The registry verifies the digest itself, so a blob that arrived corrupted is
// refused by the thing storing it rather than by the machine unpacking it a week later.
// Two steps for the blob, which is the registry's own protocol: ask for somewhere to put it, then
// put it there naming the digest. The registry verifies the digest itself, so a blob that arrived
// corrupted is refused by the thing storing it rather than by the machine unpacking it a week
// later.
//
// **Then a manifest that names it** (novox/hq issue 253, ADR 0189). The store's own collector
// marks only from manifests, and a blob no manifest names is collected however much the mesh
// means to keep it — so an archive published bare is an archive the first nightly collection
// deletes. The holder is composed from the digest and size alone (artifacts.Holder), which is
// what lets the sweep recompute it to backfill or let go without anything being recorded here.
// What a machine is told to fetch is the blob, exactly as before.
func (r Registry) PublishArchive(ctx context.Context, repository string, body []byte, digest string) (string, error) {
base := "http://" + r.Address + "/v2/" + repository
final := base + "/blobs/" + digest
// Already there. Blobs are immutable and named by their content, so this is not an
// optimisation — re-uploading would be asking the registry to store what it already has under
// the name it already has.
// the name it already has. It is still held: a blob published before holders existed is
// exactly the one a rebuild of the same source finds already there.
if there, err := r.has(ctx, final); err != nil {
return "", err
} else if there {
return final, nil
return final, r.hold(ctx, repository, digest, len(body))
}
start, err := http.NewRequestWithContext(ctx, http.MethodPost, base+"/blobs/uploads/", nil)
@@ -119,7 +131,17 @@ func (r Registry) PublishArchive(ctx context.Context, repository string, body []
said, _ := io.ReadAll(io.LimitReader(done.Body, 4096))
return "", fmt.Errorf("%s refused the blob: %s %s", base, done.Status, strings.TrimSpace(string(said)))
}
return final, nil
return final, r.hold(ctx, repository, digest, len(body))
}
// hold puts the manifest holding an archive beside it. A build whose archive could not be held is
// a failed build: recorded as published, it would be an archive the store's collector takes.
func (r Registry) hold(ctx context.Context, repository, digest string, size int) error {
store := artifacts.Store{Address: r.Address, HTTP: r.client()}
if _, err := store.HoldBlob(ctx, repository, digest, int64(size)); err != nil {
return fmt.Errorf("published %s/blobs/%s and could not hold it by a manifest: %w", repository, digest, err)
}
return nil
}
// has is whether this registry already holds what is at that URL.
+86 -6
View File
@@ -9,6 +9,8 @@ import (
"net/http/httptest"
"strings"
"testing"
"github.com/novox/mesh-controller/internal/artifacts"
)
// An OCI registry as a content-addressed blob store, which is what it is.
@@ -18,9 +20,10 @@ import (
// there is not sent again, and that a tag is never accepted as a pin.
type fakeRegistry struct {
blobs map[string][]byte
uploads int
location string
blobs map[string][]byte
manifests map[string][]byte // "<repository>@<digest>"
puts map[string]int // blob uploads, by digest
location string
}
func (f *fakeRegistry) serve(t *testing.T) *httptest.Server {
@@ -28,6 +31,8 @@ func (f *fakeRegistry) serve(t *testing.T) *httptest.Server {
if f.blobs == nil {
f.blobs = map[string][]byte{}
}
f.manifests = map[string][]byte{}
f.puts = map[string]int{}
mux := http.NewServeMux()
server := httptest.NewServer(mux)
mux.HandleFunc("/v2/", func(w http.ResponseWriter, r *http.Request) {
@@ -39,8 +44,28 @@ func (f *fakeRegistry) serve(t *testing.T) *httptest.Server {
return
}
w.WriteHeader(http.StatusNotFound)
case strings.Contains(r.URL.Path, "/manifests/"):
// The archive's holder (novox/hq issue 253): asked for with an Accept, put by digest.
repository, digest, _ := strings.Cut(strings.TrimPrefix(r.URL.Path, "/v2/"), "/manifests/")
switch r.Method {
case http.MethodHead:
if _, ok := f.manifests[repository+"@"+digest]; ok &&
strings.Contains(r.Header.Get("Accept"), "application/vnd.oci.image.manifest.v1+json") {
w.WriteHeader(http.StatusOK)
return
}
w.WriteHeader(http.StatusNotFound)
case http.MethodPut:
body, _ := io.ReadAll(r.Body)
sum := sha256.Sum256(body)
if digest != "sha256:"+hex.EncodeToString(sum[:]) {
w.WriteHeader(http.StatusBadRequest)
return
}
f.manifests[repository+"@"+digest] = body
w.WriteHeader(http.StatusCreated)
}
case r.Method == http.MethodPost && strings.HasSuffix(r.URL.Path, "/blobs/uploads/"):
f.uploads++
where := f.location
if where == "" {
where = "/v2/upload/" + hex.EncodeToString([]byte("session"))
@@ -58,6 +83,7 @@ func (f *fakeRegistry) serve(t *testing.T) *httptest.Server {
return
}
f.blobs[digest] = body
f.puts[digest]++
w.WriteHeader(http.StatusCreated)
default:
w.WriteHeader(http.StatusNotFound)
@@ -92,6 +118,57 @@ func TestAnArchiveIsStoredAndFetchableByItsDigest(t *testing.T) {
}
}
func TestAnArchiveIsPublishedWithAManifestHoldingIt(t *testing.T) {
// The store's collector marks only from manifests, so a bare blob is one the first collection
// deletes, kept or not (novox/hq issue 253, ADR 0189). Every archive goes out held, by the
// holder the sweep can compute for itself from the digest and size.
f := &fakeRegistry{}
r := registryFor(t, f)
body := []byte("a theme")
sum := sha256.Sum256(body)
digest := "sha256:" + hex.EncodeToString(sum[:])
where, err := r.PublishArchive(context.Background(), "shell/config", body, digest)
if err != nil {
t.Fatal(err)
}
if !strings.HasSuffix(where, "/v2/shell/config/blobs/"+digest) {
t.Fatalf("a machine is told to fetch %q; the blob is still what is fetched", where)
}
manifest, holder := artifacts.Holder(digest, int64(len(body)))
if got := f.manifests["shell/config@"+holder]; string(got) != string(manifest) {
t.Fatalf("the store holds %v; want the holder %s in the archive's own repository", f.manifests, holder)
}
if !strings.Contains(string(manifest), `"digest":"`+digest+`"`) {
t.Fatalf("the holder does not name the archive: %s", manifest)
}
empty := sha256.Sum256([]byte("{}"))
if _, ok := f.blobs["sha256:"+hex.EncodeToString(empty[:])]; !ok {
t.Fatal("the holder's empty config was never put, and a registry refuses a manifest without it")
}
}
func TestAnArchiveAlreadyThereIsStillHeld(t *testing.T) {
// A rebuild of the same source finds its archive already there — often one published bare,
// before holders. It is not sent again, and it is held.
f := &fakeRegistry{}
r := registryFor(t, f)
body := []byte("published bare")
sum := sha256.Sum256(body)
digest := "sha256:" + hex.EncodeToString(sum[:])
f.blobs[digest] = body
if _, err := r.PublishArchive(context.Background(), "shell/config", body, digest); err != nil {
t.Fatal(err)
}
if f.puts[digest] != 0 {
t.Fatal("a blob already there was sent again")
}
if _, holder := artifacts.Holder(digest, int64(len(body))); f.manifests["shell/config@"+holder] == nil {
t.Fatal("a blob already there was left bare")
}
}
func TestABlobAlreadyThereIsNotSentAgain(t *testing.T) {
// Not an optimisation: blobs are named by their content, so re-uploading is asking the
// registry to store what it already has under the name it already has.
@@ -107,8 +184,11 @@ func TestABlobAlreadyThereIsNotSentAgain(t *testing.T) {
if _, err := r.PublishArchive(context.Background(), "shell/config", body, digest); err != nil {
t.Fatal(err)
}
if f.uploads != 1 {
t.Fatalf("the blob was uploaded %d times", f.uploads)
if f.puts[digest] != 1 {
t.Fatalf("the blob was uploaded %d times", f.puts[digest])
}
if len(f.manifests) != 1 {
t.Fatalf("publishing twice left %d holders; want the one", len(f.manifests))
}
}
+1
View File
@@ -96,6 +96,7 @@ var ControllerVerbs = []Verb{
Input: schema(map[string]string{
"id": "a plan's id (as `plans` lists them): that plan, tier by tier",
"stop": "a plan's id: stop it — what was asked still builds, nothing further is asked",
"close": "a plan's id: close a plan that will not move again, as failed by hand (novox/hq issue 254)",
"repository": "owner/repository: the plan a merge there would produce, saving nothing (what-if); with paths or modules",
"paths": "with repository: the files the merge would change, comma-separated, from the repository's root",
"modules": "with repository: or the modules it would change, comma-separated",
+23 -7
View File
@@ -42,26 +42,37 @@ func theSeatDeclarer() catalogue.Manifest {
// A module assigned to a machine becomes a user with the authority its manifest declared — and the
// protocol of a seat declared by a *different* module, which is the whole reason a seat exists.
func TestAnAssignedModuleBecomesAUserWithWhatItDeclared(t *testing.T) {
// It declares where its account is delivered: a module with no own secret named broker can never
// be issued one, and is no user at all (novox/hq issue 195) — the case asserted below.
shop := catalogue.Manifest{
Module: "shop", Version: "1",
Emits: []string{"order.placed"}, Tools: []string{"price"},
Uses: []string{"telegram-sender"},
Uses: []string{"telegram-sender"},
OwnSecrets: catalogue.OwnSecrets{"broker": {Path: "/run/broker"}},
}
inv, ctx := aMeshWith(t, theSeatDeclarer(), shop)
quiet := catalogue.Manifest{Module: "quiet", Version: "1", Emits: []string{"thing.happened"}}
inv, ctx := aMeshWith(t, theSeatDeclarer(), shop, quiet)
if _, err := inv.AddNode(ctx, "one"); err != nil {
t.Fatal(err)
}
if _, err := inv.Assign(ctx, "one", "shop"); err != nil {
t.Fatal(err)
for _, module := range []string{"shop", "quiet"} {
if _, err := inv.Assign(ctx, "one", module); err != nil {
t.Fatal(err)
}
}
records, err := inv.BusRecords(ctx)
if err != nil {
t.Fatal(err)
}
on := records.Assigned["one"]
if len(on) != 1 || on[0].Module != "shop" {
t.Fatalf("the machine's modules read as %+v", on)
var on []broker.Declared
for _, d := range records.Assigned["one"] {
if d.Module == "shop" {
on = append(on, d)
}
}
if len(on) != 1 || on[0].NoAccount {
t.Fatalf("the machine's modules read as %+v", records.Assigned["one"])
}
if len(on[0].Uses) != 1 || on[0].Uses[0].Accepts[0] != "send" {
t.Fatalf("the seat it uses carries no protocol: %+v — so it would be granted nothing on a "+
@@ -98,6 +109,11 @@ func TestAnAssignedModuleBecomesAUserWithWhatItDeclared(t *testing.T) {
if !found {
t.Fatal("no user was derived for the assigned module")
}
for _, u := range users {
if u.Username() == "one.quiet" {
t.Fatal("a module with nowhere to read an account was made a user (novox/hq issue 195)")
}
}
}
// A machine holding a live token gets an enrolment user; one whose token is spent or expired does
+36 -5
View File
@@ -3,6 +3,7 @@ package inventory
import (
"context"
"encoding/json"
"sort"
"strings"
"github.com/novox/mesh-controller/internal/catalogue"
@@ -33,7 +34,8 @@ const KeptBuilds = 5
// - **a definition names it** — the reference appears in a module's recorded manifest, which is
// what the mesh would hand a machine now. No age limit: this is the floor;
// - **the mesh can still go back to it** — it is an artifact of one of the KeptBuilds most
// recent successful builds of its module;
// recent successful builds of a module the mesh still holds. A forgotten module keeps
// nothing beyond what a held definition names (novox/hq issue 253);
// - it was already collected, in which case there is nothing left to do.
//
// Returned in a stated order so two runs over the same records ask for the same things in the
@@ -109,12 +111,19 @@ func (i *Inventory) keptReferences(ctx context.Context) (map[string]bool, error)
return nil, err
}
// The KeptBuilds most recent successful builds of each module, whole.
// The KeptBuilds most recent successful builds of each module the mesh still holds, whole.
//
// **Only a module the mesh still holds can be gone back to** (novox/hq issue 253). "Somewhere
// to return to" is a reason about a module's releases; a module that has been forgotten has
// no releases left to return between, and its build rows stay only as history. Without the
// join every module ever built kept five builds' artifacts for ever — and once the store's
// collector runs for real, what the keep set says is what the disk holds.
recent, err := i.store.Pool().Query(ctx,
`select made from (
select made, row_number() over (partition by module order by at desc, id desc) as back
from build
where failed = '' and module is not null and module <> ''
select b.made, row_number() over (partition by b.module order by b.at desc, b.id desc) as back
from build b
join module m on m.name = b.module
where b.failed = '' and b.module is not null and b.module <> ''
) ranked where back <= $1`, KeptBuilds)
if err != nil {
return nil, err
@@ -169,6 +178,28 @@ func (i *Inventory) keptReferences(ctx context.Context) (map[string]bool, error)
return keep, nil
}
// KeptArchives is every archive the mesh keeps, in a stated order: the references the sweep must
// hold by a manifest before it lets anything go, and the ones an operator needs to read as all
// held before the store's collector is let loose (novox/hq issue 253, ADR 0189).
//
// Only references into the mesh's own store, and only blobs: an image is its own manifest, and a
// reference that is kept because nothing here can speak for it is not one the store can be asked
// about.
func (i *Inventory) KeptArchives(ctx context.Context) ([]string, error) {
keep, err := i.keptReferences(ctx)
if err != nil {
return nil, err
}
var out []string
for reference := range keep {
if path, ours := catalogue.InArtifactStore(reference); ours && strings.Contains(path, "/blobs/sha256:") {
out = append(out, reference)
}
}
sort.Strings(out)
return out, nil
}
// everyReferenceMade is every artifact reference any successful build recorded.
func (i *Inventory) everyReferenceMade(ctx context.Context) ([]string, error) {
rows, err := i.store.Pool().Query(ctx,
+96
View File
@@ -20,6 +20,19 @@ func ref(module, artifact string, n int) string {
return fmt.Sprintf("%s%s/%s@sha256:%064x", catalogue.ArtifactStoreScheme, module, artifact, n)
}
// holding registers a definition for each module that names no artifact, so the mesh holds the
// module and its recent builds are somewhere it can go back to — and nothing more.
func holding(t *testing.T, inv *Inventory, modules ...string) {
t.Helper()
for _, module := range modules {
m := catalogue.Manifest{Module: module, Version: "1"}
if err := inv.RegisterModule(context.Background(), m,
Source{Repository: "https://forge.invalid/" + module + ".git"}); err != nil {
t.Fatal(err)
}
}
}
// built records one successful build of a module publishing one image.
func built(t *testing.T, inv *Inventory, id, module string, n int) string {
t.Helper()
@@ -35,6 +48,7 @@ func built(t *testing.T, inv *Inventory, id, module string, n int) string {
func TestTheStoreKeepsTheRecentBuildsAndLetsGoOfTheRest(t *testing.T) {
inv := fresh(t)
ctx := context.Background()
holding(t, inv, "web")
// Eight builds of one module, oldest first. Five are kept — the newest, and the four a
// release that turns out wrong can be taken back to.
@@ -98,6 +112,7 @@ func TestWhatHasBeenCollectedIsNotOfferedAgain(t *testing.T) {
// time it runs, for ever — a number of requests that grows with the mesh's whole history.
inv := fresh(t)
ctx := context.Background()
holding(t, inv, "web")
for i := 1; i <= 7; i++ {
built(t, inv, fmt.Sprintf("b%02d", i), "web", i)
}
@@ -123,6 +138,7 @@ func TestWhatHasBeenCollectedIsNotOfferedAgain(t *testing.T) {
func TestAFailedBuildNamesNothingToCollectAndEachModuleIsCountedOnItsOwn(t *testing.T) {
inv := fresh(t)
ctx := context.Background()
holding(t, inv, "web", "db")
// A failed build published nothing, so it is neither kept nor collected — and it must not
// count against the module's five.
@@ -156,6 +172,7 @@ func TestAFailedBuildNamesNothingToCollectAndEachModuleIsCountedOnItsOwn(t *test
func TestAnArtifactRecordedWithAnAddressIsOfferedAsTheMeshRecordsOne(t *testing.T) {
inv := fresh(t)
ctx := context.Background()
holding(t, inv, "tools")
// The oldest build published the old way; five newer ones fill the module's five.
old := aBuild("a00", "tools", "")
@@ -190,3 +207,82 @@ func TestAnArtifactRecordedWithAnAddressIsOfferedAsTheMeshRecordsOne(t *testing.
t.Fatalf("offered %v again after collecting it", again)
}
}
// A forgotten module keeps nothing beyond what a held definition names (novox/hq issue 253).
//
// "Somewhere to go back to" is a reason about a module's releases, and a module the mesh no
// longer holds has none. Its build rows stay as history; its artifacts go — except one a module
// the mesh still holds names, which is the floor whatever built it.
func TestAForgottenModuleKeepsNothingAHeldDefinitionDoesNotName(t *testing.T) {
inv := fresh(t)
ctx := context.Background()
// Three builds of a module that was never held, or was held and then forgotten: within its
// five, and kept for that reason until now.
var gone []string
for i := 1; i <= 3; i++ {
gone = append(gone, built(t, inv, fmt.Sprintf("o%02d", i), "old", 200+i))
}
// A module the mesh holds, whose definition runs the forgotten module's newest image.
named := gone[2]
m := catalogue.Manifest{Module: "web", Version: "1", Resources: []map[string]any{{
"id": "app", "type": "container", "name": "web", "image": named,
}}}
if err := inv.RegisterModule(ctx, m, Source{Repository: "https://forge.invalid/web.git"}); err != nil {
t.Fatal(err)
}
go_, err := inv.ToCollect(ctx)
if err != nil {
t.Fatal(err)
}
if len(go_) != 2 || go_[0] != gone[0] || go_[1] != gone[1] {
t.Fatalf("offered %v; want %v — a forgotten module's builds are no release to go back to, "+
"and only what a held definition names stays", go_, gone[:2])
}
// And once the module is held again, its five are kept again.
holding(t, inv, "old")
again, err := inv.ToCollect(ctx)
if err != nil {
t.Fatal(err)
}
if len(again) != 0 {
t.Fatalf("offered %v for a module the mesh holds, within its five", again)
}
}
// The archives the mesh keeps are what the sweep holds before it lets anything go, and what an
// operator reads as all held before the store's collector is let loose (novox/hq issue 253).
func TestKeptArchivesAreTheKeptBlobsOnly(t *testing.T) {
inv := fresh(t)
ctx := context.Background()
holding(t, inv, "shell")
archive := func(n int) string {
return fmt.Sprintf("%sshell/config/blobs/sha256:%064x", catalogue.ArtifactStoreScheme, n)
}
for i := 1; i <= 6; i++ {
b := aBuild(fmt.Sprintf("s%02d", i), "shell", "")
b.Made = []Artifact{
{Name: "app", Kind: "image", Reference: ref("shell", "app", i)},
{Name: "config", Kind: "archive", Reference: archive(i)},
}
if err := inv.RecordBuild(ctx, b); err != nil {
t.Fatal(err)
}
}
kept, err := inv.KeptArchives(ctx)
if err != nil {
t.Fatal(err)
}
want := []string{archive(2), archive(3), archive(4), archive(5), archive(6)}
if len(kept) != len(want) {
t.Fatalf("kept archives %v; want the five recent ones and no images", kept)
}
for i := range want {
if kept[i] != want[i] {
t.Fatalf("kept archives %v; want %v", kept, want)
}
}
}
@@ -0,0 +1,22 @@
-- A newer plan supersedes the older open plans of the same repository and branch (novox/hq issue 254,
-- ADR 0218).
--
-- A merge produced a plan without looking at the plans still open, so two merges a few minutes apart
-- were two plans working the same modules, and a plan stuck waiting on something that would never
-- come stayed open for ever beside the newer ones. The newer plan now takes over what the older had
-- not yet built and the older is closed as `superseded` — a state of its own, so `plans` can say
-- which plan replaced it rather than reading as a failure.
--
-- `branch` is the branch the merge went into, so only a plan of the same branch is superseded. Empty
-- for every plan from before this was kept: which branch it answered is not known, and such a plan
-- is superseded by the next plan of its repository, whichever branch — nothing is lost by it, since
-- what it had not built is folded into the plan that supersedes it.
alter table release_plan add column branch text not null default '';
-- And the bus's user list the machine holding the bus was last sent, as a digest (novox/hq issue
-- 249). A module's new grants are refused by the bus until its user list says them, so that machine
-- is sent first whenever the list it would be sent differs from the one it was. Read from its whole
-- declaration, every pending change on it — a recorded upgrade the operator chose not to roll out —
-- went with every send anywhere. A digest and never the list (ADR 0043: the list is composed on each
-- push, never kept). Empty for a machine never sent one, which reads as behind once.
alter table node add column sent_bus_users text not null default '';
+20
View File
@@ -921,6 +921,26 @@ func (i *Inventory) RecordSent(ctx context.Context, node, digest string) error {
return err
}
// RecordSentBusUsers keeps a digest of the bus's user list a machine was just sent, by its name
// (novox/hq issue 249): whether the machine holding the bus must go first is whether this differs
// from the list composed now.
func (i *Inventory) RecordSentBusUsers(ctx context.Context, name, digest string) error {
_, err := i.store.Pool().Exec(ctx,
`update node set sent_bus_users = $2 where name = $1`, name, digest)
return err
}
// SentBusUsers is the digest of the bus's user list a machine was last sent, empty for none.
func (i *Inventory) SentBusUsers(ctx context.Context, name string) (string, error) {
var sent string
err := i.store.Pool().QueryRow(ctx,
`select sent_bus_users from node where name = $1`, name).Scan(&sent)
if errors.Is(err, pgx.ErrNoRows) {
return "", nil
}
return sent, err
}
// Outstanding is the digest of the declaration a machine was last sent, by its name, and empty
// for one that has never been sent anything.
//
+31 -18
View File
@@ -15,16 +15,19 @@ import (
// the store so a controller replaced mid-plan resumes it, and so `status` can say what a merge
// still waits for.
type Plan struct {
ID string `json:"id"`
Repository string `json:"repository"`
Commit string `json:"commit"`
Created time.Time `json:"created"`
Updated time.Time `json:"updated"`
State string `json:"state"`
Tier int `json:"tier"`
Tiers [][]string `json:"tiers"`
Modules map[string]*PlanModule `json:"modules"`
Note string `json:"note,omitempty"`
ID string `json:"id"`
Repository string `json:"repository"`
// Branch is the branch the merge went into (novox/hq issue 254): a newer plan supersedes the open
// ones of the same repository and branch. Empty for a plan from before it was kept.
Branch string `json:"branch,omitempty"`
Commit string `json:"commit"`
Created time.Time `json:"created"`
Updated time.Time `json:"updated"`
State string `json:"state"`
Tier int `json:"tier"`
Tiers [][]string `json:"tiers"`
Modules map[string]*PlanModule `json:"modules"`
Note string `json:"note,omitempty"`
}
// PlanModule is one module's state within a plan.
@@ -37,8 +40,15 @@ type PlanModule struct {
// later tier is built by it (ADR 0163's gate): the reports that open the gate are the ones
// after this.
SentAt *time.Time `json:"sent_at,omitempty"`
Commit string `json:"commit,omitempty"`
Why string `json:"why,omitempty"`
// First is the machines the plan sent the new build to first, and FirstAt when (novox/hq issue
// 249, ADR 0218): unless the module's policy rolls it out together, one machine takes it before
// the rest, and the rest are sent once that one reports it applied. Kept so a controller
// replaced while the plan waits on that report resumes the wait rather than sending again. The
// machine holding the bus is among them when its user list had to go first.
First []string `json:"first,omitempty"`
FirstAt *time.Time `json:"first_at,omitempty"`
Commit string `json:"commit,omitempty"`
Why string `json:"why,omitempty"`
}
// The states a plan passes through.
@@ -47,6 +57,9 @@ const (
PlanRolling = "rolling"
PlanDone = "done"
PlanFailed = "failed"
// PlanSuperseded is a plan a newer merge of the same repository and branch took over (novox/hq
// issue 254, ADR 0218): what it had not built is in the newer plan, and its note names it.
PlanSuperseded = "superseded"
)
// Open says whether the plan is still being worked.
@@ -63,11 +76,11 @@ func (i *Inventory) SavePlan(ctx context.Context, p Plan) error {
return err
}
_, err = i.store.Pool().Exec(ctx,
`insert into release_plan (id, repository, commit_hash, created, updated, state, tier, tiers, modules, note)
values ($1, $2, $3, $4, now(), $5, $6, $7, $8, $9)
`insert into release_plan (id, repository, commit_hash, created, updated, state, tier, tiers, modules, note, branch)
values ($1, $2, $3, $4, now(), $5, $6, $7, $8, $9, $10)
on conflict (id) do update set updated = now(), state = excluded.state, tier = excluded.tier,
tiers = excluded.tiers, modules = excluded.modules, note = excluded.note`,
p.ID, p.Repository, p.Commit, p.Created, p.State, p.Tier, tiers, modules, p.Note)
tiers = excluded.tiers, modules = excluded.modules, note = excluded.note, branch = excluded.branch`,
p.ID, p.Repository, p.Commit, p.Created, p.State, p.Tier, tiers, modules, p.Note, p.Branch)
return err
}
@@ -95,7 +108,7 @@ func (i *Inventory) PlanByID(ctx context.Context, id string) (Plan, error) {
func (i *Inventory) plans(ctx context.Context, tail string) ([]Plan, error) {
rows, err := i.store.Pool().Query(ctx,
`select id, repository, commit_hash, created, updated, state, tier, tiers, modules, note
`select id, repository, commit_hash, created, updated, state, tier, tiers, modules, note, branch
from release_plan `+tail)
if err != nil {
return nil, err
@@ -106,7 +119,7 @@ func (i *Inventory) plans(ctx context.Context, tail string) ([]Plan, error) {
var p Plan
var tiers, modules []byte
if err := rows.Scan(&p.ID, &p.Repository, &p.Commit, &p.Created, &p.Updated, &p.State,
&p.Tier, &tiers, &modules, &p.Note); err != nil {
&p.Tier, &tiers, &modules, &p.Note, &p.Branch); err != nil {
return nil, err
}
if err := json.Unmarshal(tiers, &p.Tiers); err != nil {
+27
View File
@@ -46,3 +46,30 @@ func TestAPlanIsKeptAdvancedAndResumedFromTheStore(t *testing.T) {
t.Fatalf("a done plan is still among the recent ones: %+v", recent)
}
}
// novox/hq issue 254: a plan keeps the branch its merge went into, and a superseded plan is not open.
func TestASupersededPlanIsNotOpen(t *testing.T) {
inv := ForTest(t)
ctx := t.Context()
p := Plan{ID: "plan-1", Repository: "novox/mesh-catalog", Branch: "main", Commit: "abc",
Created: time.Now().UTC(), State: PlanBuilding, Tiers: [][]string{{"gitea"}},
Modules: map[string]*PlanModule{"gitea": {}}}
if err := inv.SavePlan(ctx, p); err != nil {
t.Fatal(err)
}
kept, err := inv.PlanByID(ctx, "plan-1")
if err != nil || kept.Branch != "main" {
t.Fatalf("the branch was not kept: %v %+v", err, kept)
}
kept.State = PlanSuperseded
kept.Note = "superseded at tier 0 by plan-2"
if err := inv.SavePlan(ctx, kept); err != nil {
t.Fatal(err)
}
if open, err := inv.OpenPlans(ctx); err != nil || len(open) != 0 {
t.Fatalf("a superseded plan is still open: %v %+v", err, open)
}
if recent, _ := inv.RecentPlans(ctx, 5); len(recent) != 1 || recent[0].State != PlanSuperseded {
t.Fatalf("a superseded plan is not among the recent ones as superseded: %+v", recent)
}
}
+25
View File
@@ -0,0 +1,25 @@
package inventory
import "testing"
// novox/hq issue 249: the digest of the user list a machine was last sent is kept, by its name, and
// is empty for a machine never sent one.
func TestTheUserListAMachineWasSentIsKept(t *testing.T) {
inv := ForTest(t)
ctx := t.Context()
if _, err := inv.AddNode(ctx, "anchor"); err != nil {
t.Fatal(err)
}
if sent, err := inv.SentBusUsers(ctx, "anchor"); err != nil || sent != "" {
t.Fatalf("a machine never sent a list has %q: %v", sent, err)
}
if err := inv.RecordSentBusUsers(ctx, "anchor", "abc"); err != nil {
t.Fatal(err)
}
if sent, err := inv.SentBusUsers(ctx, "anchor"); err != nil || sent != "abc" {
t.Fatalf("the list sent was not kept: %q %v", sent, err)
}
if sent, err := inv.SentBusUsers(ctx, "nobody"); err != nil || sent != "" {
t.Fatalf("a machine the mesh does not know: %q %v", sent, err)
}
}
+6
View File
@@ -315,6 +315,12 @@ func TestNatsTheEventsTheControllerFollowsArriveAndAreAcknowledged(t *testing.T)
ctx, stop := context.WithCancel(context.Background())
defer stop()
go func() { _ = s.Serve(ctx) }()
// The controller's event consumer is made from now (novox/hq issue 248): what was published before
// it existed is history it never replays. So the announcements are made once it is there.
eventually(t, "the controller's event consumer being made", func() bool {
_, err := js.Context().ConsumerInfo("EVENTS", broker.ControllerName)
return err == nil
})
moved, _ := json.Marshal(Upgraded{Module: "gitea", Commit: "abcdef0123"})
if _, err := js.Context().Publish(broker.ControllerFollows[0], moved); err != nil {