diff --git a/cmd/mesh-controller/collect.go b/cmd/mesh-controller/collect.go index 6d01e2a..d30344e 100644 --- a/cmd/mesh-controller/collect.go +++ b/cmd/mesh-controller/collect.go @@ -31,7 +31,12 @@ func collect(ctx context.Context, inv *inventory.Inventory) { fmt.Fprintf(os.Stderr, "could not work out what the artifact store may let go of: %v\n", err) return } - if len(references) == 0 { + kept, err := inv.KeptArchives(ctx) + if err != nil { + fmt.Fprintf(os.Stderr, "could not work out which archives the artifact store keeps: %v\n", err) + return + } + if len(references) == 0 && len(kept) == 0 { return } shelf, err := inv.Catalogue(ctx) @@ -59,6 +64,25 @@ func collect(ctx context.Context, inv *inventory.Inventory) { defer stop() store := artifacts.Store{Address: address} + // **Hold before letting go** (novox/hq issue 253). The store's collector keeps only what a + // manifest names, and archives were published as bare blobs, so every kept archive is first + // held by its manifest — which backfills the ones published before holders, a few at a time + // as builds come, and is two HEADs each once done. A kept archive that could not be held stops + // the sweep before it deletes anything: "everything kept is held" is the precondition the + // collector's safety rests on, and a store refusing a hold would refuse the deletes too. + wrote, missing, err := holdKept(within, store, kept) + if wrote > 0 { + fmt.Fprintf(os.Stderr, "the artifact store now holds %d more kept archive(s) by a manifest\n", wrote) + } + if missing > 0 { + fmt.Fprintf(os.Stderr, "%d archive(s) the mesh keeps are not in the artifact store at all; "+ + "`collection` lists them\n", missing) + } + if err != nil { + fmt.Fprintf(os.Stderr, "not every kept archive could be held, so nothing was let go: %v\n", err) + return + } + var done []string var left, skipped int for i, reference := range references { @@ -114,6 +138,34 @@ func collect(ctx context.Context, inv *inventory.Inventory) { } } +// holdKept holds every kept archive by its manifest, stopping at the first refusal by the store. +// Answers how many holders it wrote and how many kept archives the store does not have. +// +// A missing archive is counted rather than fatal: there is nothing to hold, and that is a fact +// for an operator to read (`collection`), not a reason to stop collecting what is not kept. A +// reference the store cannot be asked about is skipped as the deletion loop skips one +// (novox/hq issue 226). +func holdKept(ctx context.Context, store artifacts.Store, kept []string) (wrote, missing int, err error) { + for _, reference := range kept { + if err := ctx.Err(); err != nil { + return wrote, missing, fmt.Errorf("ran out of time before %s: %w", reference, err) + } + did, err := store.Hold(ctx, reference) + switch { + case err == nil: + if did { + wrote++ + } + case errors.Is(err, artifacts.Gone): + missing++ + case errors.Is(err, artifacts.ErrNotOurs): + default: + return wrote, missing, fmt.Errorf("holding %s: %w", reference, err) + } + } + return wrote, missing, nil +} + // mostPerSweep is how many artifacts one sweep will ask about. Enough that a mesh building // several times a day converges within days of this landing; small enough that no single build // waits on the whole backlog. diff --git a/cmd/mesh-controller/collection.go b/cmd/mesh-controller/collection.go new file mode 100644 index 0000000..6e2520d --- /dev/null +++ b/cmd/mesh-controller/collection.go @@ -0,0 +1,166 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "flag" + "fmt" + "os" + "strings" + + "github.com/novox/mesh-controller/internal/artifacts" +) + +// What the artifact store keeps, whether each kept archive is held, and what the sweep may let go +// (novox/hq issue 253, ADR 0189). +// +// **The question to answer before the store's collector runs for real.** The collector deletes +// every blob no manifest names, and archives were published as bare blobs, so the nightly step +// runs `--dry-run` until every archive the mesh keeps is held by its manifest. The sweep holds +// them as builds come; this says how far that has got — "0 unheld" is the number that lets the +// dry run go. +// +// Reads and changes nothing: each kept archive is asked about with HEADs only. Reached over the +// console through the mesh-controller seat's `command` verb (`collection --json`), which needs no +// new verb in the seat's row. + +type collectionReport struct { + // Store is the artifact store as this machine reached it; empty when it is not on the network. + Store string `json:"store"` + // KeptArchives is how many archives the mesh keeps, for either reason. + KeptArchives int `json:"kept_archives"` + // Held is how many of them the store holds by their manifest. + Held int `json:"held"` + // Unheld are the kept archives the collector would delete tonight if it ran for real. + Unheld []string `json:"unheld"` + // Missing are kept archives the store does not have at all. + Missing []string `json:"missing"` + // Unasked is how many could not be asked about, and why the asking stopped. + Unasked int `json:"unasked"` + Stopped string `json:"stopped,omitempty"` + // Eligible is what the sweep may let go of: made by the mesh, kept for no reason, not yet + // collected — split by kind. + Eligible int `json:"eligible"` + EligibleImages int `json:"eligible_images"` + EligibleArchives int `json:"eligible_archives"` + // SafeToCollect is whether every kept archive was asked about and every one is held. + SafeToCollect bool `json:"safe_to_collect"` +} + +func collectionCommand(ctx context.Context, args []string) error { + set := flag.NewFlagSet("collection", flag.ContinueOnError) + asJSON := set.Bool("json", false, "answer as JSON") + positionals, err := parseAround(set, args) + if err != nil { + return err + } + if len(positionals) != 0 { + return errors.New("collection [--json]") + } + + open, err := openStores(ctx) + if err != nil { + return err + } + defer open.Close() + inv := open.inventory + + kept, err := inv.KeptArchives(ctx) + if err != nil { + return err + } + eligible, err := inv.ToCollect(ctx) + if err != nil { + return err + } + report := collectionReport{KeptArchives: len(kept), Eligible: len(eligible), Unheld: []string{}, Missing: []string{}} + for _, reference := range eligible { + if strings.Contains(reference, "/blobs/") { + report.EligibleArchives++ + } else { + report.EligibleImages++ + } + } + + shelf, err := inv.Catalogue(ctx) + if err != nil { + return err + } + report.Store, err = artifactStoreAddress(ctx, inv, shelf, "") + if err != nil { + return err + } + if report.Store == "" { + report.Unasked = len(kept) + report.Stopped = "this mesh has no artifact store on its network" + } else { + report.Unasked, report.Stopped = askHeld(ctx, artifacts.Store{Address: report.Store}, kept, &report) + } + report.SafeToCollect = report.Unasked == 0 && len(report.Unheld) == 0 + + if *asJSON { + encoder := json.NewEncoder(os.Stdout) + encoder.SetIndent("", " ") + return encoder.Encode(report) + } + printCollection(report) + return nil +} + +// askHeld asks the store about each kept archive, stopping at the first answer that is not about +// the archive: a store that cannot be reached for one cannot be for the next, and a page of +// identical failures says less than one line. +func askHeld(ctx context.Context, store artifacts.Store, kept []string, report *collectionReport) (int, string) { + for i, reference := range kept { + held, err := store.Held(ctx, reference) + switch { + case err == nil && held: + report.Held++ + case err == nil: + report.Unheld = append(report.Unheld, reference) + case errors.Is(err, artifacts.Gone): + report.Missing = append(report.Missing, reference) + case errors.Is(err, artifacts.ErrNotOurs): + // KeptArchives names only the mesh's own; counted as unasked if one ever is not. + report.Unasked++ + default: + return report.Unasked + len(kept) - i, fmt.Sprintf("asking about %s: %v", reference, err) + } + } + return report.Unasked, "" +} + +func printCollection(r collectionReport) { + store := r.Store + if store == "" { + store = "(not on the network)" + } + fmt.Printf("artifact store %s\n", store) + fmt.Printf("kept archives %d\n", r.KeptArchives) + fmt.Printf(" held %d\n", r.Held) + fmt.Printf(" unheld %d\n", len(r.Unheld)) + fmt.Printf(" missing %d\n", len(r.Missing)) + if r.Unasked > 0 { + fmt.Printf(" not asked %d (%s)\n", r.Unasked, r.Stopped) + } + fmt.Printf("eligible to let go %d (%d images, %d archives)\n", r.Eligible, r.EligibleImages, r.EligibleArchives) + if len(r.Unheld) > 0 { + fmt.Println("\nunheld — the store's collector would delete these; the next build's sweep holds them:") + for _, reference := range r.Unheld { + fmt.Printf(" %s\n", reference) + } + } + if len(r.Missing) > 0 { + fmt.Println("\nmissing — kept by the mesh, not in the store:") + for _, reference := range r.Missing { + fmt.Printf(" %s\n", reference) + } + } + fmt.Println() + if r.SafeToCollect { + fmt.Println("every kept archive is held: the store's collector may run for real") + } else { + fmt.Println("NOT every kept archive is known to be held: keep the store's collector on --dry-run") + } +} diff --git a/cmd/mesh-controller/delivery_order_test.go b/cmd/mesh-controller/delivery_order_test.go new file mode 100644 index 0000000..878a201 --- /dev/null +++ b/cmd/mesh-controller/delivery_order_test.go @@ -0,0 +1,145 @@ +package main + +import ( + "context" + "errors" + "reflect" + "strings" + "testing" + + "github.com/novox/mesh-controller/internal/link" +) + +// recordingDelivery is a delivery that writes down what was done, in order, and fails where told. +type recordingDelivery struct { + did []string + grantErr error + declareErr error +} + +func (r *recordingDelivery) grant(_ context.Context, sending []readyNode) error { + for _, s := range sending { + r.did = append(r.did, "grant "+s.node) + } + return r.grantErr +} + +func (r *recordingDelivery) declare(_ context.Context, s readyNode, _ []byte) (string, error) { + if r.declareErr != nil { + return "", r.declareErr + } + r.did = append(r.did, "declare "+s.node) + return "digest-" + s.node, nil +} + +func ready(names ...string) []readyNode { + var out []readyNode + for _, n := range names { + out = append(out, readyNode{node: n, declared: sendable{Resources: []map[string]any{{"id": "x"}}}}) + } + return out +} + +// novox/hq issue 249: the grants that come with a module's new declarations are issued before any +// machine is sent the code that uses them — except the machine holding the bus, whose declaration +// carries the controller's own right to issue them, and goes first. +func TestGrantsAreIssuedBeforeTheDeclarations(t *testing.T) { + d := &recordingDelivery{} + digests, err := deliver(t.Context(), d, "", ready("anchor", "laptop")) + if err != nil { + t.Fatal(err) + } + want := []string{"grant anchor", "grant laptop", "declare anchor", "declare laptop"} + if !reflect.DeepEqual(d.did, want) { + t.Fatalf("delivered in the order %v, wanted %v", d.did, want) + } + if digests["anchor"] != "digest-anchor" || digests["laptop"] != "digest-laptop" { + t.Fatalf("the digests sent were not answered: %v", digests) + } + + held := &recordingDelivery{} + if _, err := deliver(t.Context(), held, "broker", ready("broker", "anchor")); err != nil { + t.Fatal(err) + } + want = []string{"declare broker", "grant broker", "grant anchor", "declare anchor"} + if !reflect.DeepEqual(held.did, want) { + t.Fatalf("with the bus's machine in the send: %v, wanted %v", held.did, want) + } +} + +// A grant that cannot be issued holds back the machines it concerns and is an error the caller +// retries on — never "until the next push" — and the bus's own machine is sent regardless, so the +// grant that would let the controller issue memberships is never held behind them. +func TestAGrantThatFailsHoldsBackWhatItConcerns(t *testing.T) { + // A failure naming no machine (the buckets): everything but the bus's machine. + d := &recordingDelivery{grantErr: errors.New("the bus refused the bucket")} + _, err := deliver(t.Context(), d, "broker", ready("broker", "anchor", "laptop")) + if err == nil || !errors.Is(err, errGrants) || !strings.Contains(err.Error(), "the bus refused the bucket") || + !strings.Contains(err.Error(), "anchor, laptop not sent") { + t.Fatalf("a failed grant was not said as the send's failure: %v", err) + } + if want := []string{"declare broker", "grant broker", "grant anchor", "grant laptop"}; !reflect.DeepEqual(d.did, want) { + t.Fatalf("delivered %v, wanted the bus's machine alone", d.did) + } + + // A membership that failed for one machine: that machine alone. + one := &recordingDelivery{grantErr: &grantsRefused{nodes: map[string]error{"laptop": errors.New("no")}}} + _, err = deliver(t.Context(), one, "", ready("anchor", "laptop")) + if !errors.Is(err, errGrants) || !strings.Contains(err.Error(), "laptop not sent") { + t.Fatalf("one machine's refused membership was not said: %v", err) + } + if want := []string{"grant anchor", "grant laptop", "declare anchor"}; !reflect.DeepEqual(one.did, want) { + t.Fatalf("delivered %v, wanted anchor sent and laptop held back", one.did) + } + + // The announced upgrade that hit it is asked again. + if !errors.Is(askAgainOnGrants(err), link.ErrTryAgain) { + t.Fatal("an announcement whose send stopped at its grants is not asked again") + } + if other := errors.New("laptop could not be resolved"); errors.Is(askAgainOnGrants(other), link.ErrTryAgain) { + t.Fatal("any failure is asked again, not only a grant's") + } + + // Nothing to send is nothing granted either. + none := &recordingDelivery{grantErr: errors.New("never asked")} + if _, err := deliver(t.Context(), none, "", nil); err != nil || len(none.did) != 0 { + t.Fatalf("an empty send granted or failed: %v %v", none.did, err) + } +} + +// Whether the bus's machine goes first is read from the user list alone, by its digest. +func TestTheBusMachineIsBehindByItsUserListAlone(t *testing.T) { + list := "users: [a, b]" + if userListBehind(list, digestOf([]byte(list))) { + t.Fatal("the list it was sent reads as behind") + } + if !userListBehind(list, digestOf([]byte("users: [a]"))) || !userListBehind(list, "") { + t.Fatal("a changed or never-sent list reads as current") + } + if userListBehind("", "") { + t.Fatal("a machine sent no list reads as behind") + } +} + +// The machine holding the bus goes first: its declaration carries the user list the new grants are +// checked against. Among the machines it is moved to the front; not among them it is added only +// when it is behind. +func TestTheMachineHoldingTheBusIsSentFirst(t *testing.T) { + for _, c := range []struct { + what string + names []string + holder string + behind bool + want []string + }{ + {"among them", []string{"ace", "g14", "novox"}, "novox", false, []string{"novox", "ace", "g14"}}, + {"not among them, behind", []string{"ace", "g14"}, "novox", true, []string{"novox", "ace", "g14"}}, + {"not among them, current", []string{"ace", "g14"}, "novox", false, []string{"ace", "g14"}}, + {"nothing holds the bus", []string{"ace", "g14"}, "", true, []string{"ace", "g14"}}, + {"only it", []string{"novox"}, "novox", false, []string{"novox"}}, + } { + if got := brokerFirst(c.names, c.holder, c.behind); !reflect.DeepEqual(got, c.want) { + t.Errorf("%s: sent in the order %v, wanted %v", c.what, got, c.want) + } + } +} diff --git a/cmd/mesh-controller/hold_test.go b/cmd/mesh-controller/hold_test.go index 9c2b480..8301b42 100644 --- a/cmd/mesh-controller/hold_test.go +++ b/cmd/mesh-controller/hold_test.go @@ -18,16 +18,16 @@ func TestASendRoundGivesItsHoldBackOnEveryWayOut(t *testing.T) { plain := func(context.Context, string) (sendable, error) { return sendable{Resources: []map[string]any{{"id": "x"}}}, nil } - failing := func(readyNode, []byte) error { return errors.New("the broker went away") } - fine := func(readyNode, []byte) error { return nil } + failing := &recordingDelivery{declareErr: errors.New("the broker went away")} + fine := &recordingDelivery{} for name, round := range map[string]func() error{ "a body that cannot be marshalled": func() error { - _, err := sendRound(ctx, open, []string{"anchor"}, unmarshallable, fine) + _, err := sendRound(ctx, open, []string{"anchor"}, unmarshallable, fine, "", nil) return err }, "a send that fails": func() error { - _, err := sendRound(ctx, open, []string{"anchor"}, plain, failing) + _, err := sendRound(ctx, open, []string{"anchor"}, plain, failing, "", nil) return err }, } { diff --git a/cmd/mesh-controller/main.go b/cmd/mesh-controller/main.go index 0f2561f..b15f40d 100644 --- a/cmd/mesh-controller/main.go +++ b/cmd/mesh-controller/main.go @@ -73,6 +73,8 @@ func run() error { return askCommand(ctx, args[1:]) case "builds": return buildsCommand(ctx, args[1:]) + case "collection": + return collectionCommand(ctx, args[1:]) case "plans": return plansCommand(ctx, args[1:]) case "pin": @@ -200,6 +202,7 @@ func usage() { build --behind build every module the mesh holds older than its source build --on rebuild every module that stands on this module's artifacts, bases first builds [] what has been built lately, and what came of it + collection [--json] kept archives held/unheld by a manifest, and what the sweep may let go builder issue a broker account for a build machine, scoped to build work, delivered as the builder module's broker secret (module add it first) licence add|list|use|key model access, under the name a person calls it diff --git a/cmd/mesh-controller/nodes.go b/cmd/mesh-controller/nodes.go index 04a38e4..f025414 100644 --- a/cmd/mesh-controller/nodes.go +++ b/cmd/mesh-controller/nodes.go @@ -386,8 +386,12 @@ func brokerCommand(ctx context.Context, args []string) error { if len(args) > 0 && args[0] == "accounts" { return busAccounts(ctx, args[1:]) } + if len(args) > 0 && args[0] == "consumer-reset" { + return consumerReset(args[1:]) + } if len(args) == 0 || args[0] != "show" { - return errors.New("broker show | broker certificate [--check] --into | broker accounts --into ") + return errors.New("broker show | broker certificate [--check] --into | broker accounts --into | " + + "broker consumer-reset ") } known, err := broker.FromEnvironment() if errors.Is(err, broker.ErrNotConfigured) { @@ -407,6 +411,34 @@ func brokerCommand(ctx context.Context, args []string) error { return nil } +// consumerReset re-makes one consumer on a stream that keeps history to start from now (novox/hq issue +// 248): the way out of a consumer replaying a week of announcements, said rather than done by hand. A +// person's act — what was pending is dropped — so it is a command, and nothing calls it on its own. +func consumerReset(args []string) error { + if len(args) != 2 { + return errors.New("broker consumer-reset , e.g. broker consumer-reset EVENTS controller") + } + address, err := broker.BusAddress() + if err != nil { + return err + } + js, err := broker.Dial(address) + if err != nil { + return fmt.Errorf("cannot reach the bus: %w", err) + } + defer js.Close() + before, after, err := js.ResetConsumer(args[0], args[1]) + if err != nil { + return err + } + fmt.Printf("consumer %s on %s re-made to deliver from now\n", args[1], args[0]) + fmt.Printf(" before: delivers %s, delivered to %d, acknowledged to %d, %d pending, %d unacknowledged\n", + before.DeliverPolicy, before.Delivered, before.AckFloor, before.Pending, before.AckPending) + fmt.Printf(" after: delivers %s, %d pending; what was pending is dropped. A holder bound to it may need its "+ + "process restarted to bind again\n", after.DeliverPolicy, after.Pending) + return nil +} + // heardFrom says when a node was last heard from, in a form somebody can act on. // // "never" and "an hour ago" are different answers and are kept different. A node that has never diff --git a/cmd/mesh-controller/order_test.go b/cmd/mesh-controller/order_test.go index 2acac30..f4a2c62 100644 --- a/cmd/mesh-controller/order_test.go +++ b/cmd/mesh-controller/order_test.go @@ -147,6 +147,14 @@ func TestAMergeRebuildsTheModulesItChanged(t *testing.T) { {"a module the mesh does not hold", merge([]string{"modules/plex/index.ts"}, false), ""}, {"nothing said about the files", merge(nil, false), "gitea,keycloak"}, {"more files than were listed", merge([]string{"modules/gitea/index.ts"}, true), "gitea,keycloak"}, + // novox/hq issue 252: a module the mesh has never registered is still a module, when the merge + // shows it is one — and a directory that may be shared code is still shared. + {"a new module beside a held one", merge([]string{"modules/gitea/x", "modules/newmod/module.json"}, false), "gitea"}, + {"a new module's other files", merge([]string{"modules/newmod/index.ts", "modules/newmod/module.json"}, false), ""}, + {"a module removed", merge([]string{"modules/gone/module.json"}, false), ""}, + {"a directory with no manifest", merge([]string{"modules/lib/x.go"}, false), "gitea,keycloak"}, + {"a file directly among the modules", merge([]string{"modules/README.md"}, false), "gitea,keycloak"}, + {"the root's files still", merge([]string{"tsconfig.json"}, false), "gitea,keycloak"}, } { if got := named(whatTheMergeTouched(candidates, known, c.m)); got != c.want { t.Errorf("%s: rebuilt %q, wanted %q", c.what, got, c.want) diff --git a/cmd/mesh-controller/plan.go b/cmd/mesh-controller/plan.go index c3f51a8..b39fd30 100644 --- a/cmd/mesh-controller/plan.go +++ b/cmd/mesh-controller/plan.go @@ -398,7 +398,7 @@ func declarationWith(ctx context.Context, open *stores, node string, return sendable{}, err } return sendable{Resources: composed.Resources, Adoption: adoption, - Received: composed.Received, Mesh: with.Mesh, + Received: composed.Received, Mesh: with.Mesh, BusUsers: with.BusUsers, LeftOut: sortedKeysOf(composed.LeftOut), leftOutWhy: composed.LeftOut}, nil } @@ -1355,6 +1355,20 @@ func composeBusUsers(ctx context.Context, inv *inventory.Inventory, // // Asked of what this push resolves to rather than of the seat's holder mesh-wide: the file is a // resource of that module, so the question is whether it is here. + list, missing, err := busUserList(ctx, inv, onThisNode) + if len(missing) > 0 { + fmt.Printf("the bus's user list leaves out %d user(s) the mesh has minted no credential "+ + "for: %s. Each is a user that cannot connect until one is issued\n", + len(missing), strings.Join(missing, ", ")) + } + return list, err +} + +// busUserList is composeBusUsers without saying anything: the list, and the users left out of it +// for want of a credential. Asked on every send to decide whether the machine holding the bus must +// go first (novox/hq issue 249), where saying the same missing users each time would bury them. +func busUserList(ctx context.Context, inv *inventory.Inventory, + onThisNode []catalogue.Manifest) (string, []string, error) { holdsTheBus := false for _, m := range onThisNode { if m.BusUsers != "" && m.ClaimsSeat("mesh-broker") { @@ -1362,37 +1376,33 @@ func composeBusUsers(ctx context.Context, inv *inventory.Inventory, } } if !holdsTheBus { - return "", nil + return "", nil, nil } records, err := inv.BusRecords(ctx) if err != nil { - return "", err + return "", nil, err } users, err := broker.Users(records) if err != nil { - return "", err + return "", nil, err } kept, err := inv.BusUsers(ctx) if err != nil { - return "", err + return "", nil, err } hashes := make(map[string]string, len(kept)) for name, u := range kept { hashes[name] = u.PasswordHash } filled, missing := broker.WithPasswords(users, hashes) - if len(missing) > 0 { - fmt.Printf("the bus's user list leaves out %d user(s) the mesh has minted no credential "+ - "for: %s. Each is a user that cannot connect until one is issued\n", - len(missing), strings.Join(missing, ", ")) - } if len(filled) == 0 { - return "", fmt.Errorf( + return "", missing, fmt.Errorf( "this machine runs the bus and not one user has a credential, so the composed list " + "would refuse every connection in the mesh") } - return broker.ComposeAccounts(filled) + list, err := broker.ComposeAccounts(filled) + return list, missing, err } // providerModuleOf is which module answers a need on the providing node: the one in this node's diff --git a/cmd/mesh-controller/push.go b/cmd/mesh-controller/push.go index 95dea38..cc656bb 100644 --- a/cmd/mesh-controller/push.go +++ b/cmd/mesh-controller/push.go @@ -375,6 +375,17 @@ func pushCommand(ctx context.Context, args []string) error { asked = append(asked, n.Name) } + // **The machine holding the bus first** (novox/hq issue 249): its declaration carries the bus's + // user list, and a module's new grants are refused by the bus until that list says them. Among + // the machines asked it goes first; not among them and behind, it is added — a named push whose + // module gained a state would otherwise send the code and leave the right to use it for the + // cascade below, after. + holder, holderBehind, err := brokerBehind(ctx, open, asked) + if err != nil { + return err + } + asked = brokerFirst(asked, holder, holderBehind) + // Held from composing to sending, so a converge on one of them cannot send between the two // and be overtaken by what was composed before it (novox/hq ADR 0100). held, release, err := holdNodes(ctx, open, asked) @@ -404,43 +415,22 @@ func pushCommand(ctx context.Context, args []string) error { return declared, err }) - sentDigest := map[string]string{} defer release() - for _, s := range sending { - // The number is inside the signed bytes, so a replayed older declaration cannot borrow a - // newer one's (novox/hq 04-ISSUES/107); it was taken when the composition began (issue 204). - body, err := s.declared.Body() - if err != nil { - return err - } - // A running module's data moving holds this machine, and only this one (novox/hq ADR 0217). - if moves, err := heldMoves(ctx, inv, s.node, body, movable); err != nil { - return err - } else if len(moves) > 0 { - sayHeld(os.Stdout, s.node, moves) - heldBack = append(heldBack, s.node) - continue - } - if err := link.Declare(ctx, server.Bus(), ident, s.node, body, 15*time.Second); err != nil { - return err - } - // After it is away, not before. A digest recorded for something that failed to send would - // make the machine look current for a declaration it never received. - digest, err := recordSent(ctx, inv, s.node, body) - if err != nil { - return err - } - sentDigest[s.node] = digest - fmt.Printf("sent %s %d resource(s)\n", s.node, len(s.declared.Resources)) + // A machine whose declaration would move a running module's data is held, and only it (novox/hq + // ADR 0217); every other machine is delivered, its memberships first, then its declaration + // (novox/hq issue 249, ADR 0218). + bus := overTheBus{open: open, server: server, signer: ident} + toSend, err := holdingBack(ctx, inv, sending, movable, &heldBack) + if err != nil { + return err + } + sentDigest, err := deliver(ctx, bus, holder, toSend) + if err != nil { + return err } release() fmt.Printf("\n%d node(s) told\n", len(sending)) reportUnheldPushed(os.Stdout, len(args) == 1, asked, unheld) - // And each machine's memberships, as every other send does (ADR 0160): a push is the one most - // operators run, and on 2026-10-01 it was the one path that issued none. - if err := issueMemberships(ctx, open, server, sending); err != nil { - return err - } // **A named push leaves the mesh consistent, not just the machine it named** (novox/hq // issue 057, ADR 0083). Assigning a cross-node consumer mints a provision, and the PROVIDER's @@ -509,24 +499,9 @@ func pushCommand(ctx context.Context, args []string) error { } return declared, err }, - func(s readyNode, body []byte) error { + bus, holder, func(held context.Context, sending []readyNode) ([]readyNode, error) { // The cascade is a push too, and held the same way (novox/hq ADR 0217). - if moves, err := heldMoves(ctx, inv, s.node, body, movable); err != nil { - return err - } else if len(moves) > 0 { - sayHeld(os.Stdout, s.node, moves) - heldBack = append(heldBack, s.node) - return nil - } - if err := link.Declare(ctx, server.Bus(), ident, s.node, body, - 15*time.Second); err != nil { - return err - } - if _, err := recordSent(ctx, inv, s.node, body); err != nil { - return err - } - fmt.Printf("sent %s %d resource(s)\n", s.node, len(s.declared.Resources)) - return nil + return holdingBack(held, inv, sending, movable, &heldBack) }) refusals = append(refusals, refused...) if err != nil { @@ -659,10 +634,10 @@ func composeEach(names []string, allot func(node string) (int64, error), // sendRound holds the named nodes, composes each and sends each that composed, and gives the hold // back on every way out — a body that cannot be marshalled and a send that fails included // (novox/hq ADR 0100). A node that cannot be composed is a refusal, not an error: the others are -// still sent. +// still sent. Their memberships go before their declarations, as every send's do (issue 249). func sendRound(ctx context.Context, open *stores, names []string, compose func(held context.Context, node string) (sendable, error), - send func(s readyNode, body []byte) error) ([]string, error) { + d delivery, holder string, keep func(context.Context, []readyNode) ([]readyNode, error)) ([]string, error) { held, release, err := holdNodes(ctx, open, names) if err != nil { return nil, err @@ -671,18 +646,276 @@ func sendRound(ctx context.Context, open *stores, names []string, sending, refused := composeEach(names, allotting(held, open.inventory), func(node string) (sendable, error) { return compose(held, node) }) - for _, s := range sending { - body, err := s.declared.Body() - if err != nil { - return refused, err - } - if err := send(s, body); err != nil { + if keep != nil { + if sending, err = keep(held, sending); err != nil { return refused, err } } + if _, err := deliver(held, d, holder, sending); err != nil { + return refused, err + } return refused, nil } +// holdingBack leaves out each machine whose declaration would move a running module's data, says so, +// and names it among those held back (novox/hq ADR 0217); the rest go on to be delivered, grants first +// (ADR 0218). A declaration's body is the same bytes however often it is read. +func holdingBack(ctx context.Context, inv *inventory.Inventory, sending []readyNode, movable map[string]bool, + heldBack *[]string) ([]readyNode, error) { + var out []readyNode + for _, s := range sending { + body, err := s.declared.Body() + if err != nil { + return nil, err + } + moves, err := heldMoves(ctx, inv, s.node, body, movable) + if err != nil { + return nil, err + } + if len(moves) > 0 { + sayHeld(os.Stdout, s.node, moves) + *heldBack = append(*heldBack, s.node) + continue + } + out = append(out, s) + } + return out, nil +} + +// delivery is the two acts of sending machines what they should be, apart, so the order between +// them is one function's and can be read and tested there (novox/hq issue 249). +type delivery interface { + // grant issues what the machines' modules may do — each module's state raised and its + // membership issued — for every machine about to be sent. + grant(ctx context.Context, sending []readyNode) error + // declare sends one machine its declaration and records it sent, answering the digest. + declare(ctx context.Context, s readyNode, body []byte) (string, error) +} + +// errGrants marks a send that stopped because what the machines' modules may do could not be issued +// (novox/hq issue 249). Nothing about the machines is wrong; asked again, it is likely to work, so an +// announcement that hits it is held and asked again. +var errGrants = errors.New("what the machines' modules may do on the bus could not be issued, and code " + + "sent before its grants is refused there") + +// grantsRefused is a grant that failed for some machines and not others: their memberships could not +// be issued, by machine, and only those machines are held back. +type grantsRefused struct{ nodes map[string]error } + +func (g *grantsRefused) Error() string { + names := make([]string, 0, len(g.nodes)) + for n := range g.nodes { + names = append(names, n) + } + sort.Strings(names) + return fmt.Sprintf("the memberships of %s could not be issued; the first: %v", + strings.Join(names, ", "), g.nodes[names[0]]) +} + +// deliver sends the machines their declarations: **the machine holding the bus, then the grants, +// then the rest** (novox/hq issue 249). +// +// A merge gave a module a new state; its bundle reached every machine within a minute, and the +// machines' permissions on the bus did not include the state until somebody pushed by hand: the code +// arrived before the right to use it. A module that read its new state on start failed its start; the +// one that was there retried for two minutes. The memberships were issued after the declarations — +// "because the runtime it is for arrives with it" — and a membership is retained last-per-subject on +// the bus (internal/link/bus.go), so issued first it waits for the runtime that arrives after it. A +// runtime still on the old code merely holds a grant it does not use yet. +// +// **The holder's declaration before the grants, though.** The bus's user list travels in it, and the +// controller's own right to publish memberships and raise buckets is in that list (the precedent of +// issue 183): grants first, and a grant the controller is not yet allowed to make would hold the very +// declaration that allows it — a lock only a hand on the broker could open. A runtime already running +// on that machine follows a membership issued after its declaration, as it always has. +// +// **A grant that cannot be issued holds back what it concerns, and says so as an error.** It used to +// be said and passed over — "the machines keep what they derive until the next push" — which reported +// a rollout done that had delivered code its machines could not run. A membership that failed holds +// back its own machine; a failure that names no machine (the buckets) holds back every machine but the +// holder, already sent. The error carries errGrants, so the caller's rollout is not marked sent and is +// tried again. +// +// The grants are issued, not waited on: a membership is a retained message the runtime reads when it +// comes, and the bus answers its publication; nothing here waits for a runtime to have read one. +func deliver(ctx context.Context, d delivery, holder string, sending []readyNode) (map[string]string, error) { + digests := map[string]string{} + if len(sending) == 0 { + return digests, nil + } + send := func(s readyNode) error { + // The number is inside the signed bytes, so a replayed older declaration cannot borrow a + // newer one's (novox/hq 04-ISSUES/107); it was taken when the composition began (issue 204). + body, err := s.declared.Body() + if err != nil { + return err + } + digest, err := d.declare(ctx, s, body) + if err != nil { + return err + } + digests[s.node] = digest + return nil + } + var rest []readyNode + for _, s := range sending { + if holder != "" && s.node == holder { + if err := send(s); err != nil { + return digests, err + } + continue + } + rest = append(rest, s) + } + held := map[string]error{} + if err := d.grant(ctx, sending); err != nil { + var some *grantsRefused + if !errors.As(err, &some) { + var names []string + for _, s := range rest { + names = append(names, s.node) + } + if len(names) == 0 { + return digests, fmt.Errorf("%w: %w", errGrants, err) + } + return digests, fmt.Errorf("%w; %s not sent: %w", errGrants, strings.Join(names, ", "), err) + } + held = some.nodes + } + var notSent []string + for _, s := range rest { + if _, refused := held[s.node]; refused { + notSent = append(notSent, s.node) + continue + } + if err := send(s); err != nil { + return digests, err + } + } + if len(notSent) > 0 { + return digests, fmt.Errorf("%w; %s not sent: %w", errGrants, strings.Join(notSent, ", "), + &grantsRefused{nodes: held}) + } + if len(held) > 0 { + // Only the holder's own memberships failed, and it was sent before them. + return digests, fmt.Errorf("%w: %w", errGrants, &grantsRefused{nodes: held}) + } + return digests, nil +} + +// overTheBus is delivery as the mesh does it: memberships on the bus, declarations signed. +type overTheBus struct { + open *stores + server *link.Server + signer link.Signer + // indent is put before each "sent" line, for the callers whose output is nested. + indent string +} + +func (b overTheBus) grant(ctx context.Context, sending []readyNode) error { + return issueMemberships(ctx, b.open, b.server, sending) +} + +func (b overTheBus) declare(ctx context.Context, s readyNode, body []byte) (string, error) { + if err := link.Declare(ctx, b.server.Bus(), b.signer, s.node, body, 15*time.Second); err != nil { + return "", err + } + // After it is away, not before. A digest recorded for something that failed to send would make + // the machine look current for a declaration it never received. + digest, err := recordSent(ctx, b.open.inventory, s.node, body) + if err != nil { + return "", err + } + if s.declared.BusUsers != "" { + // And the user list it carried, so the next send reads whether it must go first from the + // list alone (novox/hq issue 249). On the same outliving context as the send's record. + kept, cancel := context.WithTimeout(context.WithoutCancel(ctx), 10*time.Second) + err := b.open.inventory.RecordSentBusUsers(kept, s.node, digestOf([]byte(s.declared.BusUsers))) + cancel() + if err != nil { + return "", err + } + } + fmt.Printf("%ssent %s %d resource(s)\n", b.indent, s.node, len(s.declared.Resources)) + return digest, nil +} + +// brokerFirst is the machines to send in the order a grant needs (novox/hq issue 249): the machine +// holding the bus first — its declaration carries the bus's user list (composeBusUsers), and a +// module's new permissions are refused by the bus until that list says them. Among the machines it +// is moved to the front; not among them, it is added only when it is behind. +func brokerFirst(names []string, holder string, behind bool) []string { + if holder == "" { + return names + } + present := false + rest := make([]string, 0, len(names)) + for _, n := range names { + if n == holder { + present = true + continue + } + rest = append(rest, n) + } + if !present && !behind { + return names + } + return append([]string{holder}, rest...) +} + +// brokerBehind is the machine holding the bus — the one whose declaration carries the user list — +// and, when it is not among the machines named, whether the user list it would be sent now differs +// from the one it was last sent (novox/hq issue 249). +// +// **The user list alone, not the whole declaration.** Read from the whole declaration, any change +// pending on that machine — an upgrade its policy records rather than rolls out — went with every +// send anywhere, and a module running there always put it in its first wave. A digest of the list +// last sent is kept for this (ADR 0043: the list is composed on each push, never kept itself). +func brokerBehind(ctx context.Context, open *stores, names []string) (string, bool, error) { + inv := open.inventory + holders, err := seatHolders(ctx, inv) + if err != nil { + return "", false, err + } + h, held := holders[theBrokerSeat] + if !held || h.Node == "" { + return "", false, nil + } + shelf, err := inv.Catalogue(ctx) + if err != nil { + return "", false, err + } + if m, known := shelf[h.Module]; !known || m.BusUsers == "" { + // A holder that is sent no user list carries no grant: nothing to send first. + return "", false, nil + } + for _, n := range names { + if n == h.Node { + return h.Node, false, nil + } + } + plan, _, err := planFor(ctx, open, h.Node) + if err != nil { + // It cannot be worked out: sending it would refuse the whole send, and `plan` says why. + return h.Node, false, nil + } + list, _, err := busUserList(ctx, inv, plan.Modules) + if err != nil { + return h.Node, false, nil + } + sent, err := inv.SentBusUsers(ctx, h.Node) + if err != nil { + return "", false, err + } + return h.Node, userListBehind(list, sent), nil +} + +// userListBehind is whether the user list composed now is not the one last sent, by its digest. An +// empty list composed is never behind: there is nothing for it to carry. +func userListBehind(now, sentDigest string) bool { + return now != "" && digestOf([]byte(now)) != sentDigest +} + // couldNotBeResolved is what a push ends with when some machines could not be worked out. // // **After the rest have been sent, never instead of sending them.** It is still an error, because @@ -705,23 +938,36 @@ func couldNotBeResolved(refusals []string, sent int) error { // consumer and refused on the provider would leave one end holding a credential the other has // never heard of — which is the state this whole mechanism exists to make impossible. func sendTo(ctx context.Context, open *stores, names []string) error { + _, err := sendToEach(ctx, open, names) + return err +} + +// sendToEach is sendTo, answering the machines it sent: those named, and before them the machine +// holding the bus when its user list must go first (novox/hq issue 249) — so a caller that waits for +// the machines it sent waits for that one too. +func sendToEach(ctx context.Context, open *stores, names []string) ([]string, error) { inv := open.inventory ident, err := openIdentity(ctx) if err != nil { - return err + return nil, err } defer ident.Close() gens, err := generators(ctx, open) if err != nil { - return err + return nil, err } + holder, behind, err := brokerBehind(ctx, open, names) + if err != nil { + return nil, err + } + names = brokerFirst(names, holder, behind) // Held from composing to sending (novox/hq ADR 0100); a caller that holds them already — // converge, which flips the node and then sends it — is not made to wait on itself. ctx, release, err := holdNodes(ctx, open, names) if err != nil { - return err + return nil, err } defer release() @@ -750,33 +996,29 @@ func sendTo(ctx context.Context, open *stores, names []string) error { sending = append(sending, readyNode{name, declared}) } if len(refusals) > 0 { - return fmt.Errorf("nothing was sent. %d machine(s) could not be resolved:\n\n%s", + return nil, fmt.Errorf("nothing was sent. %d machine(s) could not be resolved:\n\n%s", len(refusals), strings.Join(refusals, "\n\n")) } server, err := connectLink(ctx, nil, nil, nil) if err != nil { - return err + return nil, err } defer server.Close() - for _, s := range sending { - body, err := s.declared.Body() - if err != nil { - return err - } - if err := link.Declare(ctx, server.Bus(), ident, s.node, body, 15*time.Second); err != nil { - return err - } - if _, err := recordSent(ctx, inv, s.node, body); err != nil { - return err - } - fmt.Printf(" sent %s %d resource(s)\n", s.node, len(s.declared.Resources)) - } // And every assignment on those machines its membership (novox/hq ADR 0160): composed from the // same records the bus's accounts are, so what a runtime serves and what its account may are one - // composition. Issued after the declaration, because the runtime it is for arrives with it. - return issueMemberships(ctx, open, server, sending) + // composition. **Issued before the declarations** (novox/hq issue 249): the runtime the + // membership is for arrives with the declaration, and a membership waits for it on the bus; the + // code arriving first was refused its own state until somebody pushed. + if _, err := deliver(ctx, overTheBus{open: open, server: server, signer: ident, indent: " "}, holder, sending); err != nil { + return nil, err + } + sent := make([]string, 0, len(sending)) + for _, s := range sending { + sent = append(sent, s.node) + } + return sent, nil } // issueMemberships publishes the membership of every module on the machines just sent. @@ -797,18 +1039,23 @@ func issueMemberships(ctx context.Context, open *stores, server *link.Server, se // **Every declared state's bucket, before the memberships that name it** (novox/hq ADR 0201). The // raise at start asserts them too, but a module registered and assigned since would otherwise have // its bucket only after the control plane next restarts — found the first time a module declared - // state: its bundle asked for a bucket that did not exist. Idempotent and cheap; a failure is said - // and the push stands, as a membership's is. - if buckets, err := open.inventory.DeclaredBuckets(ctx); err != nil { - fmt.Printf(" the modules' state could not be read, so no bucket was asserted: %v\n", err) - } else if _, err := broker.RaiseBuckets(broker.OnConn(bus.Conn), buckets); err != nil { - fmt.Printf(" the modules' state could not be asserted on the bus: %v — the next push tries again\n", err) + // state: its bundle asked for a bucket that did not exist. Idempotent and cheap. + // + // **A failure here is the send's failure** (novox/hq issue 249). It was said and the push stood, + // because the declarations were already away; they are sent after this now — all but the bus's + // own machine, sent before it (deliver) — and a module whose state does not exist is a module + // that fails its start, so they are not sent and the caller tries again rather than reporting the + // rollout done. + buckets, err := open.inventory.DeclaredBuckets(ctx) + if err != nil { + return fmt.Errorf("the modules' state could not be read, so no bucket was asserted: %w", err) } - // The declarations are sent and recorded by now; a membership that cannot be issued is said - // and does not unsay them. Every runtime without one serves the shape it derives (ADR 0160), so - // the push stands, the first failure is named once, and the next push tries again. - issued, failed := 0, 0 - var first error + if _, err := broker.RaiseBuckets(broker.OnConn(bus.Conn), buckets); err != nil { + return fmt.Errorf("the modules' state could not be asserted on the bus: %w", err) + } + // Every membership is tried, and the first failure named once. + issued := 0 + refused := map[string]error{} for _, s := range sent { node := s.node for _, d := range records.Assigned[node] { @@ -829,10 +1076,9 @@ func issueMemberships(ctx context.Context, open *stores, server *link.Server, se return err } if err := bus.PublishMembership(ctx, node, d.Module, body); err != nil { - if first == nil { - first = err + if refused[node] == nil { + refused[node] = fmt.Errorf("%s: %w", d.Module, err) } - failed++ continue } issued++ @@ -841,9 +1087,10 @@ func issueMemberships(ctx context.Context, open *stores, server *link.Server, se if issued > 0 { fmt.Printf(" issued %d membership(s)\n", issued) } - if failed > 0 { - fmt.Printf(" %d membership(s) could not be issued; the first: %v — the machines keep what "+ - "they derive until the next push\n", failed, first) + if len(refused) > 0 { + // Returned, never passed over (novox/hq issue 249): the declarations of the machines they are + // for are not sent, and the rollout that asked is tried again rather than waiting for a push. + return &grantsRefused{nodes: refused} } return nil } diff --git a/cmd/mesh-controller/release_plan.go b/cmd/mesh-controller/release_plan.go index 267d2c7..499760a 100644 --- a/cmd/mesh-controller/release_plan.go +++ b/cmd/mesh-controller/release_plan.go @@ -184,6 +184,7 @@ func planOfMerge(m link.SourceMoved, moved []string, edges []inventory.Edge) inv return inventory.Plan{ ID: fmt.Sprintf("plan-%d", time.Now().UnixNano()), Repository: m.Owner + "/" + m.Repo, + Branch: m.Base, Commit: m.Commit, Created: time.Now().UTC(), State: inventory.PlanBuilding, @@ -192,6 +193,57 @@ func planOfMerge(m link.SourceMoved, moved []string, edges []inventory.Edge) inv } } +// supersededBy is what a newer plan takes over from the open plans it supersedes (novox/hq issue +// 254, ADR 0218): the modules they had not finished, and those plans closed as superseded. +// +// **A merge looked at no plan but its own.** Two merges of one repository a few minutes apart were +// two open plans asking for the same modules, each sending machines what it built; and a plan that +// would never move again — waiting on a report that could not come, at 97b1b2b — stayed open for +// ever beside the newer ones, read as work in progress by everyone who looked. The newer merge is the +// newer intent for that repository and branch, so its plan takes over: every open plan of the same +// repository and branch **created before it** — by the time the plans were made, never by comparing +// commits, which have no order of their own — gives up the modules it had not built, and those are +// planned again in the newer plan beside what the newer merge moved. +// +// "Not built" is a module not yet asked, or asked and not answered; **and a module built and not +// yet sent to its machines**, where its policy rolls it out: closed, the older plan would never send +// it, and the catalogue announces no move for a rebuild (issue 189), so the newer plan builds and +// sends it. A build the older plan asked still finishes and registers as any build does — ordered by +// when it was asked (issue 219), so the newer plan's ask, made later, is the one that stands. +// +// A plan with no branch recorded is from before branches were kept, and is superseded by the next +// plan of its repository: what it had not built is folded in, so nothing is lost by it. +func supersededBy(newer inventory.Plan, open []inventory.Plan, rollsOut func(string) bool) ([]string, []inventory.Plan) { + folded := map[string]bool{} + var closed []inventory.Plan + for _, old := range open { + if old.ID == newer.ID || !old.Open() || !strings.EqualFold(old.Repository, newer.Repository) || + (old.Branch != "" && old.Branch != newer.Branch) || !old.Created.Before(newer.Created) { + continue + } + var took []string + for name, s := range old.Modules { + if s == nil || s.State != "built" || (s.SentAt == nil && rollsOut(name)) { + folded[name] = true + took = append(took, name) + } + } + sort.Strings(took) + old.State = inventory.PlanSuperseded + old.Note = fmt.Sprintf("superseded at tier %d by %s (%s at %s)", old.Tier, newer.ID, newer.Repository, short(newer.Commit)) + if len(took) > 0 { + old.Note += "; " + strings.Join(took, ", ") + " planned there again" + } + closed = append(closed, old) + } + out := make([]string, 0, len(folded)) + for name := range folded { + out = append(out, name) + } + sort.Strings(out) + return out, closed +} + // gates is what the next tier needs running from this one: a module of the tier that a later // tier is built by — the runtime dependency — and whose policy rolls it out, must be applied by // the machines running it before the next tier is asked. A base an image stands on need only be @@ -230,6 +282,22 @@ func gates(p inventory.Plan, edges []inventory.Edge, rollsOut func(string) bool) } // applied says whether every machine running the module has reported since the module was built. +// appliedEach is applied with a moment of its own for each machine: the reports that count are the ones +// after that machine was sent the build (novox/hq issue 256). +func appliedEach(module string, since func(node string) time.Time, running []string, reports []inventory.Reported) (bool, []string) { + at := map[string]*time.Time{} + for _, r := range reports { + at[r.Node] = r.At + } + var waiting []string + for _, n := range running { + if t := at[n]; t == nil || t.Before(since(n)) { + waiting = append(waiting, n) + } + } + return len(waiting) == 0, waiting +} + func applied(module string, builtAt time.Time, running []string, reports []inventory.Reported) (bool, []string) { at := map[string]*time.Time{} for _, r := range reports { @@ -339,6 +407,10 @@ func planBuilt(ctx context.Context, open *stores, module, commit, failed string, state.Why = failed p.State = inventory.PlanFailed p.Note = fmt.Sprintf("%s failed to build in tier %d", module, p.Tier) + sayUnsent(p, func(m string) bool { + u, err := inv.UpgradeOf(ctx, m) + return err == nil && u.RollOut + }) } else { state.State = "built" state.BuiltAt = &now @@ -400,8 +472,17 @@ func advanceHeld(ctx context.Context, open *stores) { moved, err := advanceOnce(ctx, open, p, edges, rollsOut) if err != nil { fmt.Printf("%s: %v\n", p.ID, err) + // Kept in the plan, so `plans` says why it has not moved rather than the log alone; + // the state is left as it was and the step is tried again on the next tick. + p.Note = "tier " + fmt.Sprint(p.Tier) + ": " + err.Error() + " — tried again" + if err := inv.SavePlan(ctx, *p); err != nil { + fmt.Printf("%s: cannot keep the plan: %v\n", p.ID, err) + } break } + if p.State == inventory.PlanFailed { + sayUnsent(p, rollsOut) + } if err := inv.SavePlan(ctx, *p); err != nil { fmt.Printf("%s: cannot keep the plan: %v\n", p.ID, err) break @@ -470,6 +551,15 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan, // a module that packages another repository's source, keeps its commit; the catalogue announces // no move for it and its machines would keep the old image until somebody pushed (novox/hq // issue 189). A module whose policy records is built and left, as its policy says. + // + // **One machine first, unless the module's policy says together** (novox/hq issue 249, ADR + // 0218). The plan sent every machine running the module at once, and the operator's policy — + // one at a time, stopping at the first that fails, which an announced upgrade honours — was not + // read here at all: a module whose new declarations broke it broke everywhere in the same + // minute. Now the first machine is sent, the plan records it and waits for that machine's report + // after the send to say it applied what it was sent; only then are the rest sent. A first machine + // that fails or refuses stops the module's rollout and the plan with it, the rest untouched. + var pending []string for _, m := range tier { state := p.Modules[m] if state == nil || state.SentAt != nil || !rollsOut(m) { @@ -479,17 +569,64 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan, if err != nil { return false, err } + policy, err := inv.UpgradeOf(ctx, m) + if err != nil { + return false, err + } + var reports []inventory.Reported + if !policy.Together { + // Read for the choice of the first machine as well as for its report. + if reports, err = inv.LastReports(ctx); err != nil { + return false, err + } + } now := time.Now().UTC() - state.SentAt = &now - if len(running) == 0 { + step := nextRollout(*state, running, policy.Together, reports, now, planWaitBound) + switch { + case step.failed != "": + state.Why = step.failed + p.State = inventory.PlanFailed + p.Note = fmt.Sprintf("%s stopped at its first machine in tier %d: %s; %s left as it was", + m, p.Tier, step.failed, orNone(strings.Join(step.rest, ", "))) + fmt.Printf("%s: %s\n", p.ID, p.Note) + return true, nil + case step.waiting != "": + pending = append(pending, fmt.Sprintf("%s on %s, sent first at %s", m, step.waiting, state.FirstAt.Local().Format("15:04"))) + continue + case len(step.send) == 0: + // No machine runs it: nothing to send, and nothing to wait for. + state.SentAt = &now continue } - if err := sendTo(ctx, open, running); err != nil { - return false, fmt.Errorf("sending %s to %s after tier %d: %w", m, strings.Join(running, ", "), p.Tier, err) + // What sendToEach answers, not what was asked: the machine holding the bus is sent before + // the first when its user list must change (issue 249), and the plan waits for it too. + sent, err := sendToEach(ctx, open, step.send) + if err != nil { + // Not marked sent, so the next step tries again (issue 249): a grant that could not be + // issued is a send that did not happen. + return false, fmt.Errorf("sending %s to %s after tier %d: %w", m, strings.Join(step.send, ", "), p.Tier, err) } - fmt.Printf("%s: tier %d built; sent %s to %s\n", p.ID, p.Tier, m, strings.Join(running, ", ")) + if step.first { + state.First = sent + state.FirstAt = &now + p.State = inventory.PlanRolling + p.Note = fmt.Sprintf("tier %d built; sent %s to %s first", p.Tier, m, strings.Join(sent, ", ")) + fmt.Printf("%s: tier %d built; sent %s to %s first, the rest once it reports it applied\n", + p.ID, p.Tier, m, strings.Join(sent, ", ")) + return true, nil + } + state.SentAt = &now + fmt.Printf("%s: tier %d built; sent %s to %s\n", p.ID, p.Tier, m, strings.Join(sent, ", ")) return true, nil } + if len(pending) > 0 { + note := "tier " + fmt.Sprint(p.Tier) + " built; waiting for " + strings.Join(pending, "; ") + + " to report it applied before the rest are sent" + changed := p.State != inventory.PlanRolling || p.Note != note + p.State = inventory.PlanRolling + p.Note = note + return changed, nil + } // And wait for what the next tier needs running. needed := gates(*p, edges, rollsOut) if len(needed) > 0 { @@ -516,10 +653,24 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan, if state.BuiltAt != nil { since = *state.BuiltAt } - if state.SentAt != nil && state.SentAt.After(since) { - since = *state.SentAt + // **Each machine from its own send** (novox/hq issue 256). With one machine first (ADR 0218) + // a module is sent twice — the first machine, then the rest — and SentAt is the second. + // Asked of every machine, the first machine's report, made between the two sends, read as + // older than the build, and the gate waited for a report it already had, for ever. + sinceFor := func(node string) time.Time { + at := since + sent := state.SentAt + for _, n := range state.First { + if n == node { + sent = state.FirstAt + } + } + if sent != nil && sent.After(at) { + at = *sent + } + return at } - if ok, on := applied(m, since, running, reports); !ok { + if ok, on := appliedEach(m, sinceFor, running, reports); !ok { waiting = append(waiting, fmt.Sprintf("%s on %s", m, strings.Join(on, ", "))) } } @@ -540,6 +691,110 @@ func advanceOnce(ctx context.Context, open *stores, p *inventory.Plan, return true, nil } +// rolloutStep is what a plan does next with one built module's machines (novox/hq issue 249). +type rolloutStep struct { + // send is the machines to send now; first, whether they are the first machine's send. + send []string + first bool + // waiting names the first machines whose report the rest wait for. + waiting string + // failed says how a first machine did not take it; rest is what is then left alone. + failed string + rest []string +} + +// nextRollout is the next step of one module's rollout in a plan (novox/hq issue 249, ADR 0218). +// +// Together, every machine running it at once, as the policy says. Otherwise one machine first — the +// first by name among those that have reported within the bound, so a laptop that is away is not +// the one the rest wait on; the first by name when none has; the same choice on every controller and +// every resume — and the rest once each machine the first send reached reports, about the +// declaration it was last sent, that it applied it. Compared by the store's own record of what was +// sent (Reported.Current), never by this controller's clock against the machine's. +// +// **A first machine that fails, refuses, or does not report within the bound stops the rollout +// there** (ADR 0218 §2), naming the machine; the rest are not sent. A wait with no end is not a +// rollout: it held the plan open for ever, read as work in progress (issue 254). +func nextRollout(s inventory.PlanModule, running []string, together bool, reports []inventory.Reported, + now time.Time, bound time.Duration) rolloutStep { + if len(running) == 0 { + return rolloutStep{} + } + if together { + return rolloutStep{send: running} + } + byNode := map[string]inventory.Reported{} + for _, r := range reports { + byNode[r.Node] = r + } + if s.FirstAt == nil { + sorted := append([]string{}, running...) + sort.Strings(sorted) + for _, n := range sorted { + if r, said := byNode[n]; said && r.At != nil && now.Sub(*r.At) <= bound { + return rolloutStep{send: []string{n}, first: true} + } + } + return rolloutStep{send: sorted[:1], first: true} + } + sentFirst := map[string]bool{} + for _, n := range s.First { + sentFirst[n] = true + } + var rest []string + for _, n := range running { + if !sentFirst[n] { + rest = append(rest, n) + } + } + var waiting, failed []string + for _, n := range s.First { + r, said := byNode[n] + // Only a report about what it was last sent says anything about this build. + if !said || r.At == nil || !r.Current { + waiting = append(waiting, n) + continue + } + switch r.Outcome { + case inventory.OutcomeApplied: + case inventory.OutcomeFailed, inventory.OutcomeRefused: + failed = append(failed, n+" "+r.Outcome+" what it was sent") + default: + waiting = append(waiting, n) + } + } + if len(failed) > 0 { + return rolloutStep{failed: strings.Join(failed, "; "), rest: rest} + } + if len(waiting) > 0 { + if now.Sub(*s.FirstAt) > bound { + return rolloutStep{failed: fmt.Sprintf("%s did not report it applied within %s", + strings.Join(waiting, ", "), bound), rest: rest} + } + return rolloutStep{waiting: strings.Join(waiting, ", ")} + } + return rolloutStep{send: rest} +} + +// sayUnsent adds to an ended plan's note the modules it built and never sent (novox/hq issue 249). +// An announced move of a module a plan held was left to that plan; a plan that ends without +// sending it — failed elsewhere, or closed by hand — would leave its machines behind with nothing +// saying so. A module whose rollout stopped at its first machine is not among them: that stop was +// the point. Said once. +func sayUnsent(p *inventory.Plan, rollsOut func(string) bool) { + var unsent []string + for name, s := range p.Modules { + if s != nil && s.State == "built" && s.SentAt == nil && s.FirstAt == nil && rollsOut(name) { + unsent = append(unsent, name) + } + } + if len(unsent) == 0 || strings.Contains(p.Note, "built and never sent") { + return + } + sort.Strings(unsent) + p.Note += "; built and never sent: " + strings.Join(unsent, ", ") + " — `push --behind` sends them" +} + // planTicker advances open plans on a timer, for the steps outcomes alone cannot take. func planTicker(ctx context.Context, open *stores) { advancePlans(ctx, open) @@ -563,8 +818,10 @@ func planLine(p inventory.Plan, now time.Time) string { return fmt.Sprintf("%s %s done, %d tier(s)", p.Repository, short(p.Commit), len(p.Tiers)) case inventory.PlanFailed: return fmt.Sprintf("%s %s FAILED at %s: %s", p.Repository, short(p.Commit), where, p.Note) + case inventory.PlanSuperseded: + return fmt.Sprintf("%s %s %s", p.Repository, short(p.Commit), p.Note) } - since := now.Sub(p.Updated).Round(time.Minute) + since := now.Sub(p.Updated).Round(time.Second) late := "" if since > planWaitBound { late = " — LATE" @@ -693,7 +950,19 @@ func plansCommand(ctx context.Context, args []string) error { if *whatIf != "" { return planWhatIf(ctx, inv, *whatIf, splitList(*paths), splitList(*modules)) } - if len(positionals) == 2 && positionals[0] == "stop" { + // `stop`, or `close` (novox/hq issue 254): a person ending a plan that will not move again — one + // waiting on a report that cannot come — so it stops reading as work in progress. Marked failed + // with who ended it; what it asked still builds and registers. + if len(positionals) == 2 && (positionals[0] == "stop" || positionals[0] == "close") { + how := "stopped" + if positionals[0] == "close" { + how = "closed" + } + release, err := inv.HoldPlans(ctx, true) + if err != nil { + return err + } + defer release() p, err := inv.PlanByID(ctx, positionals[1]) if err != nil { return err @@ -702,25 +971,16 @@ func plansCommand(ctx context.Context, args []string) error { return fmt.Errorf("%s is already %s", p.ID, p.State) } p.State = inventory.PlanFailed - p.Note = "stopped by hand at tier " + fmt.Sprint(p.Tier) - release, err := inv.HoldPlans(ctx, true) - if err != nil { - return err - } - defer release() - if p, err = inv.PlanByID(ctx, positionals[1]); err != nil { - return err - } - if !p.Open() { - return fmt.Errorf("%s is already %s", p.ID, p.State) - } - p.State = inventory.PlanFailed - p.Note = "stopped by hand at tier " + fmt.Sprint(p.Tier) + p.Note = how + " by hand at tier " + fmt.Sprint(p.Tier) + sayUnsent(&p, func(m string) bool { + u, err := inv.UpgradeOf(ctx, m) + return err == nil && u.RollOut + }) if err := inv.SavePlan(ctx, p); err != nil { return err } - fmt.Printf("%s stopped at tier %d of %d; what was asked still builds and registers, nothing further is asked\n", - p.ID, p.Tier, len(p.Tiers)) + fmt.Printf("%s %s at tier %d of %d; what was asked still builds and registers, nothing further is asked\n", + p.ID, how, p.Tier, len(p.Tiers)) return nil } plans, err := inv.RecentPlans(ctx, *limit) @@ -795,7 +1055,13 @@ func planWhatIf(ctx context.Context, inv *inventory.Inventory, repository string how := "built; its policy records, so nothing is sent" if u, err := inv.UpgradeOf(ctx, name); err == nil && u.RollOut { running, _ := inv.Running(ctx, name) + reports, _ := inv.LastReports(ctx) how = "built, then sent to " + orNone(strings.Join(running, ", ")) + // One machine first unless the policy says together (novox/hq issue 249). + if first := nextRollout(inventory.PlanModule{}, running, u.Together, reports, time.Now(), planWaitBound); first.first && len(running) > 1 { + how = fmt.Sprintf("built, then sent to %s first and to the rest once it has applied it", + first.send[0]) + } rolls[name] = how } fmt.Printf(" %-22s %s\n", name, how) diff --git a/cmd/mesh-controller/release_plan_test.go b/cmd/mesh-controller/release_plan_test.go index 93a2288..365cdfc 100644 --- a/cmd/mesh-controller/release_plan_test.go +++ b/cmd/mesh-controller/release_plan_test.go @@ -181,3 +181,34 @@ func TestAPlanSettlesAnAskedBuildFromTheRecords(t *testing.T) { t.Errorf("a recorded failure did not fail the plan: %+v %+v", q, q.Modules["x"]) } } + +// The first machine's report, made between the send to it and the send to the rest, opens the gate for it: +// each machine is judged from its own send, not from the last one (novox/hq issue 256). +func TestTheGateJudgesEachMachineFromItsOwnSend(t *testing.T) { + built := time.Date(2026, 10, 5, 18, 22, 0, 0, time.UTC) + firstSent := built.Add(31 * time.Second) + firstReported := built.Add(43 * time.Second) + restSent := built.Add(58 * time.Second) + restReported := built.Add(74 * time.Second) + reports := []inventory.Reported{ + {Node: "ace", At: &firstReported}, + {Node: "g14", At: &restReported}, + } + since := func(node string) time.Time { + if node == "ace" { + return firstSent + } + return restSent + } + if ok, waiting := appliedEach("build-agent", since, []string{"ace", "g14"}, reports); !ok { + t.Fatalf("the gate still waits on %v, though each reported after its own send", waiting) + } + // The old reading, every machine from the last send, is what held the plan. + if ok, _ := applied("build-agent", restSent, []string{"ace", "g14"}, reports); ok { + t.Fatal("the single-moment reading should hold the first machine back") + } + early := built.Add(10 * time.Second) + if ok, waiting := appliedEach("build-agent", since, []string{"ace"}, []inventory.Reported{{Node: "ace", At: &early}}); ok || waiting[0] != "ace" { + t.Fatal("a report from before the machine was sent opened the gate") + } +} diff --git a/cmd/mesh-controller/rollout_first_test.go b/cmd/mesh-controller/rollout_first_test.go new file mode 100644 index 0000000..644f4be --- /dev/null +++ b/cmd/mesh-controller/rollout_first_test.go @@ -0,0 +1,152 @@ +package main + +import ( + "reflect" + "strings" + "testing" + "time" + + "github.com/novox/mesh-controller/internal/inventory" +) + +// novox/hq issue 249, ADR 0218: a plan rolls a module out to one machine first and the rest only +// once that machine has reported it applied; a module whose policy says together goes everywhere at +// once, as before. +func TestAPlanSendsOneMachineFirstAndTheRestAfterItsReport(t *testing.T) { + running := []string{"novox", "ace", "g14"} + sentAt := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC) + now := sentAt.Add(5 * time.Minute) + bound := 30 * time.Minute + after := sentAt.Add(time.Minute) + next := func(s inventory.PlanModule, running []string, together bool, reports []inventory.Reported) rolloutStep { + return nextRollout(s, running, together, reports, now, bound) + } + + // Together: every machine at once. + if step := next(inventory.PlanModule{}, running, true, nil); !reflect.DeepEqual(step.send, running) || step.first { + t.Fatalf("a together policy did not send every machine at once: %+v", step) + } + + // Otherwise the first by name, alone, when none has reported lately. + step := next(inventory.PlanModule{}, running, false, nil) + if !step.first || !reflect.DeepEqual(step.send, []string{"ace"}) { + t.Fatalf("the first send was %+v, wanted ace alone", step) + } + + state := inventory.PlanModule{First: []string{"ace"}, FirstAt: &sentAt} + report := func(outcome string, current bool) []inventory.Reported { + return []inventory.Reported{{Node: "ace", At: &after, Outcome: outcome, Current: current}, + {Node: "g14", At: &after, Outcome: inventory.OutcomeApplied, Current: true}} + } + + // No report yet, or one about an older declaration than it was last sent: wait. + for what, reports := range map[string][]inventory.Reported{ + "no report": nil, + "a report about older": report(inventory.OutcomeApplied, false), + } { + step := next(state, running, false, reports) + if len(step.send) != 0 || step.waiting != "ace" || step.failed != "" { + t.Errorf("%s: %+v, wanted to wait for ace", what, step) + } + } + + // Applied what it was last sent: the rest, and only the rest. + step = next(state, running, false, report(inventory.OutcomeApplied, true)) + if step.first || !reflect.DeepEqual(step.send, []string{"novox", "g14"}) { + t.Fatalf("after ace applied it the plan sent %+v, wanted novox and g14", step) + } + + // Failed or refused: stop, the rest untouched. + for _, outcome := range []string{inventory.OutcomeFailed, inventory.OutcomeRefused} { + step := next(state, running, false, report(outcome, true)) + if len(step.send) != 0 || !strings.Contains(step.failed, "ace "+outcome) || + !reflect.DeepEqual(step.rest, []string{"novox", "g14"}) { + t.Errorf("a first machine that %s it: %+v", outcome, step) + } + } + + // The machine holding the bus went with the first send: the rest wait for it too, and it is + // not sent again. + both := inventory.PlanModule{First: []string{"novox", "ace"}, FirstAt: &sentAt} + half := report(inventory.OutcomeApplied, true) + if step := next(both, running, false, half); step.waiting != "novox" { + t.Fatalf("the plan did not wait for the bus's machine sent first: %+v", step) + } + all := append(half, inventory.Reported{Node: "novox", At: &after, Outcome: inventory.OutcomeApplied, Current: true}) + if step := next(both, running, false, all); !reflect.DeepEqual(step.send, []string{"g14"}) { + t.Fatalf("after both applied it the plan sent %+v, wanted g14 alone", step) + } + + // One machine, or none: nothing is waited for that cannot come. + if step := next(inventory.PlanModule{}, nil, false, nil); len(step.send) != 0 || step.first { + t.Fatalf("a module nothing runs was sent: %+v", step) + } + if step := next(state, []string{"ace"}, false, report(inventory.OutcomeApplied, true)); len(step.send) != 0 || step.waiting != "" { + t.Fatalf("a module on one machine waited for more: %+v", step) + } +} + +// ADR 0218 §2: a first machine that does not report within the bound stops the rollout there, +// naming the machine and the bound; the rest are left alone. +func TestAFirstMachineThatDoesNotReportStopsTheRollout(t *testing.T) { + sentAt := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC) + state := inventory.PlanModule{First: []string{"ace"}, FirstAt: &sentAt} + step := nextRollout(state, []string{"ace", "g14"}, false, nil, sentAt.Add(31*time.Minute), 30*time.Minute) + if !strings.Contains(step.failed, "ace did not report it applied within 30m") || + !reflect.DeepEqual(step.rest, []string{"g14"}) || len(step.send) != 0 { + t.Fatalf("a silent first machine: %+v", step) + } +} + +// The first machine is the first by name among those heard from lately: a laptop that is away is +// not the one the rest wait on. When none has been heard from, the first by name. +func TestTheFirstMachineIsOneThatHasReportedLately(t *testing.T) { + now := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC) + lately, long := now.Add(-time.Minute), now.Add(-3*time.Hour) + reports := []inventory.Reported{ + {Node: "ace", At: &long}, {Node: "g14", At: &lately}, {Node: "novox", At: &lately}, + } + step := nextRollout(inventory.PlanModule{}, []string{"novox", "ace", "g14"}, false, reports, now, 30*time.Minute) + if !step.first || !reflect.DeepEqual(step.send, []string{"g14"}) { + t.Fatalf("the first send was %+v, wanted g14, the first heard from lately", step) + } +} + +// An announced move of a module an open plan is still rolling out is left to the plan: sending it +// here as well put the bundle on every machine at once (novox/hq issue 249). +func TestAnAnnouncedMoveIsLeftToThePlanRollingItOut(t *testing.T) { + sent := time.Now() + plans := []inventory.Plan{ + {ID: "plan-done", State: inventory.PlanDone, Modules: map[string]*inventory.PlanModule{"agent": {}}}, + {ID: "plan-1", State: inventory.PlanRolling, Modules: map[string]*inventory.PlanModule{ + "agent": {State: "built", First: []string{"ace"}, FirstAt: &sent}, "gitea": {State: "built", SentAt: &sent}}}, + } + if got := rolledOutByAPlan(plans, "agent"); got != "plan-1" { + t.Fatalf("a module the plan is rolling out was not left to it: %q", got) + } + for _, m := range []string{"gitea", "keycloak"} { + if got := rolledOutByAPlan(plans, m); got != "" { + t.Errorf("%s, which no plan will send, was left to %s", m, got) + } + } +} + +// A plan that ends without sending what it built says so, with the remedy; a module whose rollout +// stopped at its first machine is not among them. +func TestAnEndedPlanSaysWhatItBuiltAndNeverSent(t *testing.T) { + at := time.Now() + p := inventory.Plan{State: inventory.PlanFailed, Note: "closed by hand at tier 1", + Modules: map[string]*inventory.PlanModule{ + "agent": {State: "built"}, + "stopped": {State: "built", First: []string{"ace"}, FirstAt: &at}, + "sent": {State: "built", SentAt: &at}, + "notes": {State: "built"}, + "later": {}, + }} + rollsOut := func(m string) bool { return m != "notes" } + sayUnsent(&p, rollsOut) + sayUnsent(&p, rollsOut) + if p.Note != "closed by hand at tier 1; built and never sent: agent — `push --behind` sends them" { + t.Fatalf("the note reads %q", p.Note) + } +} diff --git a/cmd/mesh-controller/seatverbs.go b/cmd/mesh-controller/seatverbs.go index 96666b5..2f8dfe3 100644 --- a/cmd/mesh-controller/seatverbs.go +++ b/cmd/mesh-controller/seatverbs.go @@ -100,6 +100,9 @@ func argvFor(verb string, args map[string]any) ([]string, error) { if id := str("stop"); id != "" { return []string{"plans", "stop", id}, nil } + if id := str("close"); id != "" { + return []string{"plans", "close", id}, nil + } if id := str("id"); id != "" { return []string{"plans", id}, nil } @@ -214,7 +217,7 @@ func argvFor(verb string, args map[string]any) ([]string, error) { } // jsonVerbs are the verbs whose command speaks JSON, so the answer carries it as data as well. -var jsonVerbs = map[string]bool{"status": true, "seats": true, "plan": true} +var jsonVerbs = map[string]bool{"status": true, "seats": true, "plan": true, "collection": true} // runVerb runs this binary with the given command line and gathers what it said. func runVerb(ctx context.Context, argv []string) (verbAnswer, error) { diff --git a/cmd/mesh-controller/sendable.go b/cmd/mesh-controller/sendable.go index 6722732..4219e97 100644 --- a/cmd/mesh-controller/sendable.go +++ b/cmd/mesh-controller/sendable.go @@ -31,6 +31,10 @@ type sendable struct { // the same composition as its received files, and every machine's private-network address. Received map[string]map[string][]catalogue.Contribution Mesh []string + // BusUsers is the bus's user list this declaration carries, empty for every machine but the one + // holding the bus; not sent apart from the file it is in. Its digest is recorded once sent, so + // whether that machine must go first is read from the list alone (novox/hq issue 249). + BusUsers string // LeftOut is every module of the machine's set left out of this declaration because a stored // setting cannot compose with its definition (novox/hq ADR 0163, rule 6), sorted. The host // keeps that module's held things and touches none of its containers; a machine is told diff --git a/cmd/mesh-controller/supersede_test.go b/cmd/mesh-controller/supersede_test.go new file mode 100644 index 0000000..2c269fb --- /dev/null +++ b/cmd/mesh-controller/supersede_test.go @@ -0,0 +1,95 @@ +package main + +import ( + "reflect" + "strings" + "testing" + "time" + + "github.com/novox/mesh-controller/internal/inventory" +) + +// novox/hq issue 254, ADR 0218: a newer plan takes over what the older open plans of its repository +// and branch had not built, and closes them as superseded; another repository's plan, another +// branch's, and a plan made after it are left alone. +func TestANewerPlanSupersedesTheOlderOpenPlansOfItsRepository(t *testing.T) { + at := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC) + sent := at.Add(time.Minute) + plan := func(id, repository, branch string, created time.Time, modules map[string]*inventory.PlanModule) inventory.Plan { + return inventory.Plan{ID: id, Repository: repository, Branch: branch, Commit: id + "-commit", + Created: created, State: inventory.PlanRolling, Modules: modules} + } + older := plan("plan-1", "novox/mesh-catalog", "main", at, map[string]*inventory.PlanModule{ + "gitea": {State: "built", SentAt: &sent}, // done with: stays done + "keycloak": {State: "asked"}, // asked, not answered: folded + "plex": {}, // not yet asked: folded + "agent": {State: "built"}, // built, rolls out, not sent: folded + "notes": {State: "built"}, // built, records: nothing to send + }) + stuck := plan("plan-0", "Novox/Mesh-Catalog", "", at.Add(-time.Hour), map[string]*inventory.PlanModule{ + "runtime": {State: "asked"}, + }) + other := plan("plan-2", "novox/mesh-controller", "main", at, map[string]*inventory.PlanModule{"mesh-controller": {}}) + release := plan("plan-3", "novox/mesh-catalog", "release", at, map[string]*inventory.PlanModule{"lemurs": {}}) + later := plan("plan-5", "novox/mesh-catalog", "main", at.Add(2*time.Hour), map[string]*inventory.PlanModule{"later": {}}) + done := plan("plan-6", "novox/mesh-catalog", "main", at, map[string]*inventory.PlanModule{"finished": {}}) + done.State = inventory.PlanDone + + newer := plan("plan-4", "novox/mesh-catalog", "main", at.Add(time.Hour), nil) + newer.Commit = "97b1b2b0c0ffee" + rollsOut := func(m string) bool { return m != "notes" } + folded, closed := supersededBy(newer, []inventory.Plan{stuck, older, other, release, later, done, newer}, rollsOut) + + if want := []string{"agent", "keycloak", "plex", "runtime"}; !reflect.DeepEqual(folded, want) { + t.Fatalf("folded %v, wanted %v", folded, want) + } + var ids []string + for _, p := range closed { + ids = append(ids, p.ID) + if p.State != inventory.PlanSuperseded || p.Open() { + t.Errorf("%s was left %s", p.ID, p.State) + } + if !strings.Contains(p.Note, "plan-4") || !strings.Contains(p.Note, "97b1b2b0") { + t.Errorf("%s does not name the plan that superseded it: %q", p.ID, p.Note) + } + } + if want := []string{"plan-0", "plan-1"}; !reflect.DeepEqual(ids, want) { + t.Fatalf("superseded %v, wanted %v — another repository, another branch, a later plan and a "+ + "finished one are left alone", ids, want) + } + if other.State != inventory.PlanRolling { + t.Fatal("the plan handed in was changed in place") + } + if line := planLine(closed[1], time.Now()); !strings.Contains(line, "superseded") { + t.Fatalf("a superseded plan reads %q", line) + } +} + +// novox/hq issue 254: a person closes a plan that will not move again, by its id. +func TestAPersonClosesAStuckPlan(t *testing.T) { + open := aMesh(t) + ctx := t.Context() + stuck := inventory.Plan{ID: "plan-97b1b2b", Repository: "novox/mesh-catalog", Commit: "97b1b2b", + Created: time.Now().UTC(), State: inventory.PlanRolling, Tier: 1, Tiers: [][]string{{"a"}, {"b"}}, + Modules: map[string]*inventory.PlanModule{"a": {State: "built"}, "b": {}}} + if err := open.inventory.SavePlan(ctx, stuck); err != nil { + t.Fatal(err) + } + if err := plansCommand(ctx, []string{"close", stuck.ID}); err != nil { + t.Fatal(err) + } + closed, err := open.inventory.PlanByID(ctx, stuck.ID) + if err != nil { + t.Fatal(err) + } + if closed.State != inventory.PlanFailed || !strings.Contains(closed.Note, "closed by hand") { + t.Fatalf("the plan was left %s: %q", closed.State, closed.Note) + } + if err := plansCommand(ctx, []string{"close", stuck.ID}); err == nil { + t.Fatal("a plan already closed was closed again") + } + if argv, err := argvFor("plans", map[string]any{"close": stuck.ID}); err != nil || + !reflect.DeepEqual(argv, []string{"plans", "close", stuck.ID}) { + t.Fatalf("the seat's verb does not close a plan: %v %v", argv, err) + } +} diff --git a/cmd/mesh-controller/upgrades.go b/cmd/mesh-controller/upgrades.go index b16542b..3fa7aab 100644 --- a/cmd/mesh-controller/upgrades.go +++ b/cmd/mesh-controller/upgrades.go @@ -5,6 +5,7 @@ import ( "errors" "flag" "fmt" + "path" "regexp" "strings" "time" @@ -55,10 +56,26 @@ func (f following) Upgraded(ctx context.Context, u link.Upgraded) error { return nil } + // **A plan that holds the module rolls it out, and this does not** (novox/hq issue 249, ADR + // 0218). A merge's plan builds the module and sends it one machine first, the rest once that one + // has applied it; this announcement arrives as the build registers, and sending here too — one + // machine after another without waiting for any to apply — put the new bundle on every machine in + // the same minute, whatever the plan was waiting for. A move no plan answers (a build asked by + // hand) is still this handler's. + if plans, err := inv.OpenPlans(ctx); err != nil { + return notNow(err) + } else if id := rolledOutByAPlan(plans, u.Module); id != "" { + // Said with its remedy: a plan that ends without sending it — failed, or closed by hand — leaves + // these machines behind, which `status` lists and `push --behind` sends (novox/hq issue 249). + fmt.Printf("%s moved to %s; %s rolls it out to %s — if that plan ends without sending it, "+ + "`status` lists them as behind and `push --behind` sends it\n", + u.Module, shortCommit(u.Commit), id, readableList(on)) + return nil + } if decision.Together { fmt.Printf("%s moved to %s; sending %s together\n", u.Module, shortCommit(u.Commit), readableList(on)) - return sendTo(ctx, f.open, on) + return askAgainOnGrants(sendTo(ctx, f.open, on)) } // One at a time, and stopping at the first that fails. // @@ -69,13 +86,34 @@ func (f following) Upgraded(ctx context.Context, u link.Upgraded) error { u.Module, shortCommit(u.Commit), readableList(on)) for _, node := range on { if err := sendTo(ctx, f.open, []string{node}); err != nil { - return fmt.Errorf("%s did not take %s, so the machines after it were left alone: %w", - node, u.Module, err) + return askAgainOnGrants(fmt.Errorf("%s did not take %s, so the machines after it were left alone: %w", + node, u.Module, err)) } } return nil } +// askAgainOnGrants marks a send that stopped at its grants as one to ask again (novox/hq issue 249): +// an announcement handled by a send whose memberships could not be issued is held and redelivered, +// rather than taken as handled with the machines left on the old version. +func askAgainOnGrants(err error) error { + if err != nil && errors.Is(err, errGrants) && !errors.Is(err, link.ErrTryAgain) { + return fmt.Errorf("%w: %w", link.ErrTryAgain, err) + } + return err +} + +// rolledOutByAPlan is the open plan that will send a module's machines its new build — one holding +// the module that has not finished sending it — or empty when none will (novox/hq issue 249). +func rolledOutByAPlan(plans []inventory.Plan, module string) string { + for _, p := range plans { + if s, holds := p.Modules[module]; p.Open() && holds && (s == nil || (s.SentAt == nil && s.State != "failed")) { + return p.ID + } + } + return "" +} + // readableList names machines the way a sentence does, because this is read by a person deciding // whether an upgrade went where they expected. func readableList(names []string) string { @@ -323,6 +361,45 @@ func (f following) SourceMoved(ctx context.Context, m link.SourceMoved) error { } defer release() plan := planOfMerge(m, movedNames, edges) + // **A newer plan supersedes the older open plans of this repository and branch** (novox/hq issue + // 254, ADR 0218): what they had not built is planned here again, and they are closed, so one plan + // works a repository's modules at a time and a stuck one ends at the next merge. + working, err := inv.OpenPlans(ctx) + if err != nil { + return notNow(err) + } + rollsOut := func(module string) bool { + u, err := inv.UpgradeOf(ctx, module) + return err == nil && u.RollOut + } + folded, superseded := supersededBy(plan, working, rollsOut) + if len(folded) > 0 { + held := map[string]bool{} + for _, e := range entries { + held[e.Manifest.Module] = true + } + names := map[string]bool{} + for _, name := range movedNames { + names[name] = true + } + var also []string + for _, name := range folded { + // One the catalogue no longer holds would fail the newer plan's ask; it is not this + // merge's to build. + if held[name] && !names[name] { + names[name] = true + movedNames = append(movedNames, name) + also = append(also, name) + } + } + if len(also) > 0 { + again := planOfMerge(m, movedNames, edges) + again.ID, again.Created = plan.ID, plan.Created + plan = again + fmt.Printf(" %s, left unbuilt by an older plan of %s, are planned here again\n", + strings.Join(also, ", "), plan.Repository) + } + } if hasCycle(plan.Tiers, edges) { fmt.Printf(" the last tier depends on itself: %s — built together, in no order\n", strings.Join(plan.Tiers[len(plan.Tiers)-1], ", ")) @@ -330,6 +407,14 @@ func (f following) SourceMoved(ctx context.Context, m link.SourceMoved) error { if err := inv.SavePlan(ctx, plan); err != nil { return notNow(err) } + // Closed after the newer plan is kept, never before: a controller replaced between the two leaves + // both open, which the next merge settles, rather than neither. + for _, old := range superseded { + if err := inv.SavePlan(ctx, old); err != nil { + return notNow(err) + } + fmt.Printf(" %s (%s at %s) is %s\n", old.ID, old.Repository, short(old.Commit), old.Note) + } var tiers []string for i, t := range plan.Tiers { tiers = append(tiers, fmt.Sprintf("%d: %s", i, strings.Join(t, ", "))) @@ -423,25 +508,45 @@ func lastLookAt(entries []inventory.Entry, m link.SourceMoved) time.Time { // // A change inside *another* module's directory is that module's business and not this one's, even // when the mesh does not hold that module: `known` is every module this repository is known to hold, -// whatever branch it was registered from. That is also the limit of this — a repository whose shared -// code sits inside a directory the mesh has never seen a module in reads as shared, and everything -// is rebuilt. Rebuilding too much is the safe direction: the fault this whole path exists for is a -// mesh that believes it is current and is not (novox/hq 04-ISSUES/131). +// whatever branch it was registered from. +// +// **And a module the mesh has never seen is still a module** (novox/hq issue 252). A merge adding a +// new module to the catalogue repository — `modules/newmod/module.json` and its files — read as a +// change to shared code, because `modules/newmod` was nobody's known directory, and every module +// built from the repository was rebuilt and rolled out for a module none of them is. So the +// directories that hold modules are known too: the parents of the known modules' directories +// (`modules`, never the root). A changed path `//…` belongs to the module at +// `/` — held or not — and rebuilds nothing else, **provided it is shown to be a +// module**: its `module.json` is among the changed files (added, changed, or removed with it). A +// directory under the same parent whose manifest the merge did not touch may as well be a shared +// library (`modules/lib`), and that is still read as shared. Rebuilding too much remains the safe +// direction: the fault this whole path exists for is a mesh that believes it is current and is not +// (novox/hq 04-ISSUES/131). A file at the root, or directly in a parent, is shared as it always was. func whatTheMergeTouched(candidates, known []inventory.Entry, m link.SourceMoved) []inventory.Entry { // Nothing said about the files, or not all of them said: everything built from it is affected. if len(m.Paths) == 0 || m.PathsTruncated { return candidates } var dirs []string + parents := map[string]bool{} for _, e := range known { if e.Source.Path != "" && sameRepository(e.Source.Repository, m) { - dirs = append(dirs, e.Source.Path) + dir := strings.Trim(e.Source.Path, "/") + dirs = append(dirs, dir) + if parent := path.Dir(dir); parent != "." && parent != "/" { + parents[parent] = true + } } } + changed := map[string]bool{} for _, p := range m.Paths { - if !insideAny(p, dirs) { - return candidates + changed[strings.TrimPrefix(p, "/")] = true + } + for _, p := range m.Paths { + if insideAny(p, dirs) || inAModuleOfItsOwn(p, parents, changed) { + continue } + return candidates } var out []inventory.Entry for _, e := range candidates { @@ -452,6 +557,28 @@ func whatTheMergeTouched(candidates, known []inventory.Entry, m link.SourceMoved return out } +// inAModuleOfItsOwn is whether a changed file is inside a module directory the mesh does not know — +// `//…` under a directory known to hold modules, whose `module.json` the same merge +// changed (novox/hq issue 252). Such a file is that module's business and nobody else's. +func inAModuleOfItsOwn(p string, parents, changed map[string]bool) bool { + p = strings.TrimPrefix(p, "/") + for parent := range parents { + rest, under := strings.CutPrefix(p, parent+"/") + if !under { + continue + } + name, _, inADirectory := strings.Cut(rest, "/") + if !inADirectory || name == "" { + // A file directly in the parent — `modules/README.md` — is about all of them. + continue + } + if changed[parent+"/"+name+"/module.json"] { + return true + } + } + return false +} + // inside is whether a changed file is in a directory: that directory itself, or under it. func inside(path, dir string) bool { dir = strings.Trim(dir, "/") diff --git a/internal/artifacts/hold.go b/internal/artifacts/hold.go new file mode 100644 index 0000000..d8d1d9b --- /dev/null +++ b/internal/artifacts/hold.go @@ -0,0 +1,345 @@ +package artifacts + +import ( + "bytes" + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "fmt" + "io" + "net/http" + "strconv" + "strings" + "time" + + "github.com/novox/mesh-controller/internal/catalogue" +) + +// Every archive the store keeps is held by a manifest (novox/hq issue 253, ADR 0189). +// +// **The store's collector marks only from manifests.** The mesh's store is a stock registry, and +// its nightly `registry garbage-collect` walks every manifest in every repository, marks the blobs +// those manifests name, and deletes every blob it did not mark. An image is a manifest, so what +// the mesh keeps of an image survives. An archive was not: the builder put it in the store as a +// bare blob — upload, then `PUT ?digest=` — and nothing in the store names it. To the collector a +// bare blob is unreferenced, so the first real collection would have deleted every archive the +// mesh holds, kept or not, and every machine pinning a bundle would have found it gone. The +// collector runs `--dry-run` until this is true. +// +// **So each archive gets a holder**: the smallest OCI image manifest that names it — the empty +// config, one layer, nothing else — put in the archive's own repository, by digest, untagged. The +// collector marks it and so keeps the archive; the sweep lets go of an archive by deleting its +// holder first, which is what lets the bytes go at the next collection. +// +// **Nothing a machine reads changes.** The recorded reference stays +// `artifact-store:////blobs/sha256:…`, and machines fetch the blob exactly as +// before. The holder is the store's bookkeeping, not a second way to reach anything. +// +// **Deterministic, so it never needs recording.** The holder is composed from the archive's digest +// and size alone, in a fixed field order with no timestamps or annotations, so the sweep can +// compute which manifest holds any archive from the reference it already has plus one HEAD for the +// size. No schema change, no second record that could disagree with the store. + +const ( + // mediaManifest is the type a holder is put and asked for as. + mediaManifest = "application/vnd.oci.image.manifest.v1+json" + // mediaEmpty is the OCI empty descriptor's type: a config that says nothing, for a manifest + // whose only purpose is to name its layer. + mediaEmpty = "application/vnd.oci.empty.v1+json" + // emptyDigest is the digest of `{}`, the empty config's content, fixed by the OCI spec. + emptyDigest = "sha256:44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a" + // mediaArchive is the layer type an archive is held as. Every archive the builder publishes is + // `pack`'s gzipped tar, so this is the true type and not a placeholder — and it is a constant, + // not read from anywhere, because the holder must be recomputable from the reference alone. + mediaArchive = "application/vnd.oci.image.layer.v1.tar+gzip" +) + +// emptyConfig is the content emptyDigest names. +var emptyConfig = []byte("{}") + +// manifestAccept is what a manifest is asked for as. A registry answers a manifest HEAD only in a +// type the caller named, and answers 404 to a bare one for a manifest it holds perfectly well +// (measured 2026-09-28; internal/builder/registry.go says how that was found). +var manifestAccept = []string{ + mediaManifest, + "application/vnd.docker.distribution.manifest.v2+json", +} + +type descriptor struct { + MediaType string `json:"mediaType"` + Digest string `json:"digest"` + Size int64 `json:"size"` +} + +type holderManifest struct { + SchemaVersion int `json:"schemaVersion"` + MediaType string `json:"mediaType"` + Config descriptor `json:"config"` + Layers []descriptor `json:"layers"` +} + +// Holder is the manifest that holds an archive in the store, and its digest. +// +// A pure function of the archive's digest and size: the same two in give the same bytes out, +// always, because `encoding/json` writes a struct's fields in their declared order and there is +// nothing here that varies by when or where it was composed. +func Holder(digest string, size int64) (body []byte, holder string) { + body, err := json.Marshal(holderManifest{ + SchemaVersion: 2, + MediaType: mediaManifest, + Config: descriptor{MediaType: mediaEmpty, Digest: emptyDigest, Size: int64(len(emptyConfig))}, + Layers: []descriptor{{MediaType: mediaArchive, Digest: digest, Size: size}}, + }) + if err != nil { + // Marshalling a struct of strings and integers cannot fail. + panic(err) + } + sum := sha256.Sum256(body) + return body, "sha256:" + hex.EncodeToString(sum[:]) +} + +// Hold makes sure the store holds this archive by a manifest, and says whether it had to write one. +// +// Takes a reference as the mesh records it. An image is its own manifest and needs no holder, so +// it answers false and nothing is asked. Idempotent: a holder already there is left alone, which +// is what lets the sweep run it over every kept archive on every build and so backfill the bare +// blobs published before holders existed (novox/hq issue 253). +// +// Gone when the store does not hold the archive at all: there is nothing to hold, and that is a +// fact the caller reports rather than one this invents a remedy for. +func (s Store) Hold(ctx context.Context, reference string) (bool, error) { + repository, digest, archive, err := s.archive(reference) + if err != nil || !archive { + return false, err + } + size, err := s.blobSize(ctx, repository, digest) + if err != nil { + return false, err + } + return s.HoldBlob(ctx, repository, digest, size) +} + +// Held is whether the store holds this archive by its manifest. Asks and changes nothing — the +// question an operator needs answered with "none unheld" before the collector is let loose. +// +// An image answers true: it is its own manifest. An archive the store does not have answers Gone. +func (s Store) Held(ctx context.Context, reference string) (bool, error) { + repository, digest, archive, err := s.archive(reference) + if err != nil { + return false, err + } + if !archive { + return true, nil + } + size, err := s.blobSize(ctx, repository, digest) + if err != nil { + return false, err + } + _, holder := Holder(digest, size) + return s.has(ctx, s.url(repository, "manifests", holder), manifestAccept...) +} + +// HoldBlob puts the holder for a blob of this digest and size into its repository, unless it is +// there already. Answers whether it wrote one. +// +// The builder calls this with the size it has just uploaded; the sweep, through Hold, with the size +// the store reports. Both arrive at the same holder, which is the point of composing it. +func (s Store) HoldBlob(ctx context.Context, repository, digest string, size int64) (bool, error) { + if s.Address == "" { + return false, fmt.Errorf("this mesh has no artifact store on its network to hold %s/%s in", repository, digest) + } + body, holder := Holder(digest, size) + there, err := s.has(ctx, s.url(repository, "manifests", holder), manifestAccept...) + if err != nil { + return false, err + } + if there { + return false, nil + } + + // The config must be in the repository before a manifest naming it is accepted: a registry + // refuses a manifest whose blobs it cannot find there, which is the property that makes a + // holder mean something. + if err := s.putBlob(ctx, repository, emptyDigest, emptyConfig); err != nil { + return false, err + } + + // **By digest, never by tag.** A tag would be one more name to move and one more thing the + // collector's `--delete-untagged` would read as meaningful; the mesh names nothing by tag that + // it pins by digest, and an untagged manifest is kept by plain collection. + request, err := http.NewRequestWithContext(ctx, http.MethodPut, + s.url(repository, "manifests", holder), bytes.NewReader(body)) + if err != nil { + return false, err + } + request.Header.Set("Content-Type", mediaManifest) + response, err := s.client().Do(request) + if err != nil { + return false, err + } + defer response.Body.Close() + if response.StatusCode != http.StatusCreated { + said, _ := io.ReadAll(io.LimitReader(response.Body, 4096)) + return false, fmt.Errorf("the artifact store refused to hold %s/%s: %s %s", + repository, digest, response.Status, strings.TrimSpace(string(said))) + } + return true, nil +} + +// letGoOfHolder deletes the manifest holding an archive, before the archive's own link goes. +// +// **Holder first.** Deleting the blob link alone leaves a manifest still naming the blob, and the +// collector would keep its bytes for ever on the strength of it — the sweep would record the +// archive collected while the disk said otherwise. Deleting the holder first and failing before +// the link goes leaves an unheld archive that the next sweep still offers, which is safe. +// +// A store that no longer has the blob answers Gone: without its size the holder cannot be named, +// and without the blob there is nothing left for a holder to keep. A store that never had a holder +// for it — an archive published before holders, never backfilled — answers 404 to the delete, and +// that is the outcome wanted. +func (s Store) letGoOfHolder(ctx context.Context, repository, digest string) error { + size, err := s.blobSize(ctx, repository, digest) + if err != nil { + return err + } + _, holder := Holder(digest, size) + err = s.remove(ctx, s.url(repository, "manifests", holder), repository+"/manifests/"+holder) + if err == Gone { + return nil + } + return err +} + +// archive reads a recorded reference into its repository and digest, and whether it is an archive +// at all. Refuses as ErrNotOurs anything the mesh did not put in its own store. +func (s Store) archive(reference string) (repository, digest string, archive bool, err error) { + path, kept := catalogue.InArtifactStore(reference) + if !kept { + return "", "", false, fmt.Errorf("%w: %s", ErrNotOurs, reference) + } + if s.Address == "" { + return "", "", false, fmt.Errorf("this mesh has no artifact store on its network to ask about %s", reference) + } + repository, kind, digest, err := split(path) + if err != nil { + return "", "", false, err + } + return repository, digest, kind == "blobs", nil +} + +// blobSize is how large the store says a blob is; Gone when it does not have it. +func (s Store) blobSize(ctx context.Context, repository, digest string) (int64, error) { + request, err := http.NewRequestWithContext(ctx, http.MethodHead, s.url(repository, "blobs", digest), nil) + if err != nil { + return 0, err + } + response, err := s.client().Do(request) + if err != nil { + return 0, fmt.Errorf("cannot reach the artifact store at %s: %w", s.Address, err) + } + defer response.Body.Close() + switch response.StatusCode { + case http.StatusOK: + case http.StatusNotFound: + return 0, Gone + default: + return 0, fmt.Errorf("the artifact store answered %s for %s/blobs/%s", response.Status, repository, digest) + } + // Read from the header rather than ContentLength: a HEAD's ContentLength is what the response + // says it would have sent, which Go reports faithfully, but a proxy in between is free to drop + // it, and the header is what the registry itself wrote. + if length := response.Header.Get("Content-Length"); length != "" { + if n, err := strconv.ParseInt(length, 10, 64); err == nil && n >= 0 { + return n, nil + } + } + if response.ContentLength >= 0 { + return response.ContentLength, nil + } + return 0, fmt.Errorf("the artifact store holds %s/blobs/%s and will not say how large it is", repository, digest) +} + +// putBlob uploads a small blob unless the repository already has it: ask where, then put it there +// naming the digest — the registry's own two steps, the same the builder takes for an archive. +func (s Store) putBlob(ctx context.Context, repository, digest string, body []byte) error { + if there, err := s.has(ctx, s.url(repository, "blobs", digest)); err != nil { + return err + } else if there { + return nil + } + start, err := http.NewRequestWithContext(ctx, http.MethodPost, + "http://"+s.Address+"/v2/"+repository+"/blobs/uploads/", nil) + if err != nil { + return err + } + begun, err := s.client().Do(start) + if err != nil { + return fmt.Errorf("cannot start an upload to %s: %w", repository, err) + } + begun.Body.Close() + if begun.StatusCode != http.StatusAccepted { + return fmt.Errorf("the artifact store answered %s when asked where to put a blob in %s", begun.Status, repository) + } + where := begun.Header.Get("Location") + if where == "" { + return fmt.Errorf("the artifact store accepted an upload to %s and said nowhere to put it", repository) + } + if strings.HasPrefix(where, "/") { + where = "http://" + s.Address + where + } + separator := "?" + if strings.Contains(where, "?") { + separator = "&" + } + put, err := http.NewRequestWithContext(ctx, http.MethodPut, where+separator+"digest="+digest, bytes.NewReader(body)) + if err != nil { + return err + } + put.Header.Set("Content-Type", "application/octet-stream") + done, err := s.client().Do(put) + if err != nil { + return err + } + defer done.Body.Close() + if done.StatusCode != http.StatusCreated { + said, _ := io.ReadAll(io.LimitReader(done.Body, 4096)) + return fmt.Errorf("the artifact store refused a blob in %s: %s %s", repository, done.Status, strings.TrimSpace(string(said))) + } + return nil +} + +// has is whether the store answers 200 for a HEAD at that URL. +func (s Store) has(ctx context.Context, url string, accept ...string) (bool, error) { + request, err := http.NewRequestWithContext(ctx, http.MethodHead, url, nil) + if err != nil { + return false, err + } + for _, media := range accept { + request.Header.Add("Accept", media) + } + response, err := s.client().Do(request) + if err != nil { + return false, fmt.Errorf("cannot reach the artifact store at %s: %w", s.Address, err) + } + defer response.Body.Close() + switch response.StatusCode { + case http.StatusOK: + return true, nil + case http.StatusNotFound: + return false, nil + default: + return false, fmt.Errorf("the artifact store answered %s for %s", response.Status, url) + } +} + +func (s Store) url(repository, kind, digest string) string { + return "http://" + s.Address + "/v2/" + repository + "/" + kind + "/" + digest +} + +func (s Store) client() *http.Client { + if s.HTTP != nil { + return s.HTTP + } + return &http.Client{Timeout: 30 * time.Second} +} diff --git a/internal/artifacts/hold_test.go b/internal/artifacts/hold_test.go new file mode 100644 index 0000000..8f241f6 --- /dev/null +++ b/internal/artifacts/hold_test.go @@ -0,0 +1,268 @@ +package artifacts + +import ( + "bytes" + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io" + "net/http" + "net/http/httptest" + "strconv" + "strings" + "sync" + "testing" + + "github.com/novox/mesh-controller/internal/catalogue" +) + +// Every kept archive is held by a manifest (novox/hq issue 253, ADR 0189). +// +// Against an in-memory registry that keeps blobs and manifests per repository and refuses what a +// registry refuses — a blob whose digest does not match, a manifest whose digest does not match +// or whose blobs the repository does not have, a manifest asked for without an Accept naming its +// type. What is asserted is this side's decisions; the live test below asserts the registry's. + +type memRegistry struct { + mu sync.Mutex + blobs map[string][]byte // repository + "@" + digest + manifests map[string][]byte // repository + "@" + digest + writes []string // every PUT and DELETE, as "METHOD path" +} + +func digestOf(body []byte) string { + sum := sha256.Sum256(body) + return "sha256:" + hex.EncodeToString(sum[:]) +} + +func (m *memRegistry) serve(t *testing.T) Store { + t.Helper() + m.blobs = map[string][]byte{} + m.manifests = map[string][]byte{} + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + m.mu.Lock() + defer m.mu.Unlock() + path := strings.TrimPrefix(r.URL.Path, "/v2/") + if r.Method == http.MethodPut || r.Method == http.MethodDelete { + m.writes = append(m.writes, r.Method+" "+r.URL.Path) + } + switch { + case r.Method == http.MethodPost && strings.HasSuffix(path, "/blobs/uploads/"): + repository := strings.TrimSuffix(path, "/blobs/uploads/") + w.Header().Set("Location", "/upload/"+repository+"?state=x") + w.WriteHeader(http.StatusAccepted) + case r.Method == http.MethodPut && strings.HasPrefix(r.URL.Path, "/upload/"): + repository := strings.TrimPrefix(r.URL.Path, "/upload/") + body, _ := io.ReadAll(r.Body) + digest := r.URL.Query().Get("digest") + if digest != digestOf(body) { + w.WriteHeader(http.StatusBadRequest) + return + } + m.blobs[repository+"@"+digest] = body + w.WriteHeader(http.StatusCreated) + case strings.Contains(path, "/blobs/"): + repository, digest, _ := strings.Cut(path, "/blobs/") + key := repository + "@" + digest + body, ok := m.blobs[key] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + switch r.Method { + case http.MethodHead: + w.Header().Set("Content-Length", strconv.Itoa(len(body))) + w.WriteHeader(http.StatusOK) + case http.MethodDelete: + delete(m.blobs, key) + w.WriteHeader(http.StatusAccepted) + default: + w.WriteHeader(http.StatusMethodNotAllowed) + } + case strings.Contains(path, "/manifests/"): + repository, digest, _ := strings.Cut(path, "/manifests/") + key := repository + "@" + digest + switch r.Method { + case http.MethodHead: + if _, ok := m.manifests[key]; !ok || !strings.Contains(r.Header.Get("Accept"), mediaManifest) { + w.WriteHeader(http.StatusNotFound) + return + } + w.WriteHeader(http.StatusOK) + case http.MethodPut: + body, _ := io.ReadAll(r.Body) + if digest != digestOf(body) { + w.WriteHeader(http.StatusBadRequest) + return + } + var named holderManifest + if err := json.Unmarshal(body, &named); err != nil { + w.WriteHeader(http.StatusBadRequest) + return + } + for _, d := range append([]descriptor{named.Config}, named.Layers...) { + if _, ok := m.blobs[repository+"@"+d.Digest]; !ok { + w.WriteHeader(http.StatusBadRequest) + fmt.Fprintf(w, "MANIFEST_BLOB_UNKNOWN %s", d.Digest) + return + } + } + m.manifests[key] = body + w.WriteHeader(http.StatusCreated) + case http.MethodDelete: + if _, ok := m.manifests[key]; !ok { + w.WriteHeader(http.StatusNotFound) + return + } + delete(m.manifests, key) + w.WriteHeader(http.StatusAccepted) + } + default: + w.WriteHeader(http.StatusNotFound) + } + })) + t.Cleanup(server.Close) + return Store{Address: strings.TrimPrefix(server.URL, "http://")} +} + +// bare puts an archive in the store the way the builder did before holders: a blob, nothing more. +func (m *memRegistry) bare(repository string, body []byte) string { + m.mu.Lock() + defer m.mu.Unlock() + digest := digestOf(body) + m.blobs[repository+"@"+digest] = body + return catalogue.ArtifactStoreScheme + repository + "/blobs/" + digest +} + +func TestTheHolderIsComposedFromTheDigestAndSizeAlone(t *testing.T) { + // The sweep must arrive at the very manifest the builder wrote, with nothing recorded between + // them. Same inputs, same bytes — and a different size is a different holder, so a holder can + // never be mistaken for one of a different blob. + digest := "sha256:" + strings.Repeat("a", 64) + one, first := Holder(digest, 42) + two, second := Holder(digest, 42) + if !bytes.Equal(one, two) || first != second { + t.Fatalf("the same archive composed two holders:\n%s\n%s", one, two) + } + if _, other := Holder(digest, 43); other == first { + t.Fatal("a different size composed the same holder") + } + want := `{"schemaVersion":2,"mediaType":"application/vnd.oci.image.manifest.v1+json",` + + `"config":{"mediaType":"application/vnd.oci.empty.v1+json",` + + `"digest":"sha256:44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a","size":2},` + + `"layers":[{"mediaType":"application/vnd.oci.image.layer.v1.tar+gzip","digest":"` + digest + `","size":42}]}` + if string(one) != want { + t.Fatalf("the holder is\n%s\nwant\n%s", one, want) + } + if digestOf(emptyConfig) != emptyDigest { + t.Fatalf("the empty config's digest is %s, not %s", digestOf(emptyConfig), emptyDigest) + } +} + +func TestHoldBackfillsABareArchiveAndIsIdempotent(t *testing.T) { + // The archives published before this have no holder. Hold, run over every kept archive on + // every sweep, writes one the first time and nothing after. + m := &memRegistry{} + store := m.serve(t) + ctx := context.Background() + body := []byte("a theme") + reference := m.bare("shell/config", body) + + if held, err := store.Held(ctx, reference); err != nil || held { + t.Fatalf("a bare blob reads as held=%v (%v)", held, err) + } + wrote, err := store.Hold(ctx, reference) + if err != nil { + t.Fatal(err) + } + if !wrote { + t.Fatal("holding a bare archive wrote nothing") + } + _, holder := Holder(digestOf(body), int64(len(body))) + if _, ok := m.manifests["shell/config@"+holder]; !ok { + t.Fatalf("the store holds manifests %v; want %s", m.manifests, holder) + } + if held, err := store.Held(ctx, reference); err != nil || !held { + t.Fatalf("after holding, held=%v (%v)", held, err) + } + + writes := len(m.writes) + wrote, err = store.Hold(ctx, reference) + if err != nil { + t.Fatal(err) + } + if wrote || len(m.writes) != writes { + t.Fatalf("holding again wrote %v", m.writes[writes:]) + } +} + +func TestAnImageNeedsNoHolderAndAMissingArchiveIsGone(t *testing.T) { + m := &memRegistry{} + store := m.serve(t) + ctx := context.Background() + + // An image is its own manifest: nothing is asked. + wrote, err := store.Hold(ctx, catalogue.ArtifactStoreScheme+"web/app@sha256:"+strings.Repeat("b", 64)) + if err != nil || wrote || len(m.writes) != 0 { + t.Fatalf("holding an image wrote=%v err=%v writes=%v", wrote, err, m.writes) + } + // An archive the store does not have is a fact to report, not something to invent a holder for. + _, err = store.Hold(ctx, catalogue.ArtifactStoreScheme+"web/config/blobs/sha256:"+strings.Repeat("c", 64)) + if !errors.Is(err, Gone) { + t.Fatalf("holding a missing archive answered %v, want Gone", err) + } + // And a reference that is not the mesh's is refused as such. + if _, err := store.Hold(ctx, "docker.io/library/registry@sha256:abc"); !errors.Is(err, ErrNotOurs) { + t.Fatalf("holding a vendor's image answered %v, want ErrNotOurs", err) + } +} + +func TestLettingGoOfAnArchiveDeletesItsHolderFirst(t *testing.T) { + // A holder left behind would keep the bytes through every collection while the record said + // collected; the link deleted first and the holder failing after would be that exactly. + m := &memRegistry{} + store := m.serve(t) + ctx := context.Background() + body := []byte("an old theme") + reference := m.bare("shell/config", body) + if _, err := store.Hold(ctx, reference); err != nil { + t.Fatal(err) + } + m.writes = nil + + if err := store.LetGo(ctx, reference); err != nil { + t.Fatal(err) + } + _, holder := Holder(digestOf(body), int64(len(body))) + want := []string{ + "DELETE /v2/shell/config/manifests/" + holder, + "DELETE /v2/shell/config/blobs/" + digestOf(body), + } + if strings.Join(m.writes, "\n") != strings.Join(want, "\n") { + t.Fatalf("the store was asked\n%s\nwant\n%s", strings.Join(m.writes, "\n"), strings.Join(want, "\n")) + } + if len(m.manifests) != 0 { + t.Fatalf("a holder survived: %v", m.manifests) + } + // Asked again, the archive is already gone, which is the outcome wanted. + if err := store.LetGo(ctx, reference); !errors.Is(err, Gone) { + t.Fatalf("letting go twice answered %v, want Gone", err) + } +} + +func TestLettingGoOfAnUnheldArchiveStillDeletesIt(t *testing.T) { + // An archive published before holders and let go of before any sweep held it: the holder's + // delete answers 404, which is the outcome wanted, and the blob still goes. + m := &memRegistry{} + store := m.serve(t) + reference := m.bare("shell/config", []byte("never held")) + if err := store.LetGo(context.Background(), reference); err != nil { + t.Fatal(err) + } + if len(m.blobs) != 0 { + t.Fatalf("the blob survived: %v", m.blobs) + } +} diff --git a/internal/artifacts/live_registry_test.go b/internal/artifacts/live_registry_test.go new file mode 100644 index 0000000..c88b37f --- /dev/null +++ b/internal/artifacts/live_registry_test.go @@ -0,0 +1,130 @@ +package artifacts + +import ( + "bytes" + "context" + "fmt" + "io" + "net/http" + "os" + "os/exec" + "strings" + "testing" + "time" + + "github.com/novox/mesh-controller/internal/catalogue" +) + +// The registry's own collector keeps a held archive and takes a bare one (novox/hq issue 253, +// ADR 0189). +// +// Everything else here is asserted against a fake, which can only say what this side asks. This is +// the one question a fake cannot answer — what `registry garbage-collect` actually does with what +// this side wrote — and it is the whole of whether the store's nightly step may stop being a dry +// run. Against the very image the mesh's store runs: +// +// docker run -d --rm --name mesh-controller-registry -p 15000:5000 \ +// -e REGISTRY_STORAGE_DELETE_ENABLED=true registry:2.8.3 +// MESH_TEST_REGISTRY=127.0.0.1:15000 MESH_TEST_REGISTRY_CONTAINER=mesh-controller-registry \ +// go test -run Live ./internal/artifacts/ +// docker stop mesh-controller-registry +// +// Skipped without both variables: it needs a registry it may write to and collect, and a container +// to run the collector in. +func TestLiveTheRegistrysCollectorKeepsWhatIsHeldAndTakesWhatIsNot(t *testing.T) { + address := os.Getenv("MESH_TEST_REGISTRY") + container := os.Getenv("MESH_TEST_REGISTRY_CONTAINER") + if address == "" || container == "" { + t.Skip("no MESH_TEST_REGISTRY / MESH_TEST_REGISTRY_CONTAINER; see this test's comment for the registry to raise") + } + ctx := context.Background() + store := Store{Address: address} + run := time.Now().UnixNano() + + // Four archives in four repositories, each a different story. Distinct bytes per run, so a + // registry reused across runs cannot answer for an earlier one. + put := func(name string) (repository, digest string, body []byte) { + repository = fmt.Sprintf("live-%d/%s", run, name) + body = []byte(fmt.Sprintf("%s archive of run %d", name, run)) + digest = digestOf(body) + if err := store.putBlob(ctx, repository, digest, body); err != nil { + t.Fatal(err) + } + return repository, digest, body + } + reference := func(repository, digest string) string { + return catalogue.ArtifactStoreScheme + repository + "/blobs/" + digest + } + + // Published held — what PublishArchive now does. + heldRepo, heldDigest, heldBody := put("held") + if _, err := store.HoldBlob(ctx, heldRepo, heldDigest, int64(len(heldBody))); err != nil { + t.Fatalf("the registry refused a holder: %v", err) + } + // Published bare, as before, and never held: what the collector must take. + _, bareDigest, _ := put("bare") + // Published bare and then held by the sweep: the backfill. + backRepo, backDigest, backBody := put("backfilled") + if wrote, err := store.Hold(ctx, reference(backRepo, backDigest)); err != nil || !wrote { + t.Fatalf("backfilling wrote=%v: %v", wrote, err) + } + if held, err := store.Held(ctx, reference(backRepo, backDigest)); err != nil || !held { + t.Fatalf("after backfilling, held=%v: %v", held, err) + } + // Held, and then let go of by the sweep: holder first, then the link. + goneRepo, goneDigest, _ := put("let-go") + if _, err := store.Hold(ctx, reference(goneRepo, goneDigest)); err != nil { + t.Fatal(err) + } + if err := store.LetGo(ctx, reference(goneRepo, goneDigest)); err != nil { + t.Fatalf("letting go of a held archive: %v", err) + } + + collected, err := exec.CommandContext(ctx, "docker", "exec", container, + "registry", "garbage-collect", "/etc/docker/registry/config.yml").CombinedOutput() + if err != nil { + t.Fatalf("the collector failed: %v\n%s", err, collected) + } + t.Logf("the collector said:\n%s", lastLines(string(collected), 12)) + + // What is asserted is the bytes on the store's disk, not what the running server answers: the + // server caches blob descriptors in memory and can answer for a blob the collector removed. + onDisk := func(digest string) bool { + hex := strings.TrimPrefix(digest, "sha256:") + path := "/var/lib/registry/docker/registry/v2/blobs/sha256/" + hex[:2] + "/" + hex + "/data" + return exec.CommandContext(ctx, "docker", "exec", container, "test", "-f", path).Run() == nil + } + if !onDisk(heldDigest) { + t.Error("the collector took an archive published held") + } + if !onDisk(backDigest) { + t.Error("the collector took an archive the sweep backfilled a holder for") + } + if onDisk(bareDigest) { + t.Error("the collector kept a bare archive — then the holders prove nothing, and this test is wrong") + } + if onDisk(goneDigest) { + t.Error("the collector kept an archive the sweep let go of: its holder outlived its link") + } + // And what survived is still fetched exactly as machines fetch it: the blob, by digest. + for repository, want := range map[string][]byte{heldRepo: heldBody, backRepo: backBody} { + digest := digestOf(want) + response, err := http.Get(catalogue.Routed(reference(repository, digest), address)) + if err != nil { + t.Fatal(err) + } + got, _ := io.ReadAll(response.Body) + response.Body.Close() + if response.StatusCode != http.StatusOK || !bytes.Equal(got, want) { + t.Errorf("%s answered %s with %q after collection", repository, response.Status, got) + } + } +} + +func lastLines(s string, n int) string { + lines := strings.Split(strings.TrimSpace(s), "\n") + if len(lines) > n { + lines = lines[len(lines)-n:] + } + return strings.Join(lines, "\n") +} diff --git a/internal/artifacts/store.go b/internal/artifacts/store.go index ca80e65..688b933 100644 --- a/internal/artifacts/store.go +++ b/internal/artifacts/store.go @@ -1,7 +1,8 @@ // Package artifacts speaks to the mesh's artifact store over its own door. // // Only what the mesh needs that nothing else does: letting go of something it put there -// (novox/hq ADR 0189, issue 108). Pushing is the builder's, through the container runtime; reading +// (novox/hq ADR 0189, issue 108), and holding every archive it keeps by a manifest so the store's +// own collector does not take it (novox/hq issue 253). Pushing is the builder's, through the container runtime; reading // is every machine's, through its runtime. This is the one operation that belongs to the thing // holding the records, because it is the only one that is a decision rather than a transfer. package artifacts @@ -12,7 +13,6 @@ import ( "fmt" "net/http" "strings" - "time" "github.com/novox/mesh-controller/internal/catalogue" ) @@ -43,6 +43,9 @@ var ErrNotOurs = errors.New("not a reference into the mesh's artifact store") // an image, `…/blobs/sha256:…` for an archive — because that is the identity every record uses, // and composes the address here at the moment of use. // +// An archive is let go of in two deletes, its holder manifest and then the blob's link (novox/hq +// issue 253); an image in one. +// // Returns Gone when the store answers that it does not have it. That is not a failure: the sweep // wants the artifact absent, and it is. It is distinguished from success only so a caller can say // which of the two happened. @@ -66,17 +69,24 @@ func (s Store) LetGo(ctx context.Context, reference string) error { if err != nil { return err } - url := "http://" + s.Address + "/v2/" + repository + "/" + kind + "/" + digest + if kind == "blobs" { + // **An archive's holder goes before the archive** (novox/hq issue 253): a manifest left + // naming the blob would keep its bytes through every collection while the record said + // collected. Gone here means the store has no such blob, so there is nothing to let go. + if err := s.letGoOfHolder(ctx, repository, digest); err != nil { + return err + } + } + return s.remove(ctx, s.url(repository, kind, digest), reference) +} +// remove asks the store to delete what is at url. Gone when it has no such thing. +func (s Store) remove(ctx context.Context, url, what string) error { request, err := http.NewRequestWithContext(ctx, http.MethodDelete, url, nil) if err != nil { return err } - client := s.HTTP - if client == nil { - client = &http.Client{Timeout: 30 * time.Second} - } - response, err := client.Do(request) + response, err := s.client().Do(request) if err != nil { return err } @@ -92,9 +102,9 @@ func (s Store) LetGo(ctx context.Context, reference string) error { return fmt.Errorf( "the artifact store refuses deletion: its server was started without it enabled "+ "(REGISTRY_STORAGE_DELETE_ENABLED), so nothing can be collected until the store "+ - "module is applied again (novox/hq ADR 0189). Asking about %s", reference) + "module is applied again (novox/hq ADR 0189). Asking about %s", what) default: - return fmt.Errorf("the artifact store answered %s for %s", response.Status, reference) + return fmt.Errorf("the artifact store answered %s for %s", response.Status, what) } } diff --git a/internal/artifacts/store_test.go b/internal/artifacts/store_test.go index ee09086..097b9a1 100644 --- a/internal/artifacts/store_test.go +++ b/internal/artifacts/store_test.go @@ -20,6 +20,17 @@ func fakeStore(t *testing.T, answer int) (Store, *[]string) { t.Helper() var asked []string server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Method == http.MethodHead && strings.Contains(r.URL.Path, "/blobs/") { + // An archive's size, asked so its holder can be named (novox/hq issue 253). A store + // that does not have the thing does not have its blob either. + if answer == http.StatusNotFound { + w.WriteHeader(http.StatusNotFound) + return + } + w.Header().Set("Content-Length", "7") + w.WriteHeader(http.StatusOK) + return + } if r.Method != http.MethodDelete { t.Errorf("the store was asked %s %s; collecting is a delete", r.Method, r.URL.Path) } @@ -45,8 +56,14 @@ func TestAnImageAndAnArchiveAreAskedForAtTheirOwnEndpoints(t *testing.T) { if err := store.LetGo(ctx, archive); err != nil { t.Fatal(err) } - want := []string{"/v2/web/app/manifests/sha256:abc123", "/v2/web/config/blobs/sha256:def456"} - if len(*asked) != 2 || (*asked)[0] != want[0] || (*asked)[1] != want[1] { + // The archive's holder goes first, then the archive (novox/hq issue 253). + _, holder := Holder("sha256:def456", 7) + want := []string{ + "/v2/web/app/manifests/sha256:abc123", + "/v2/web/config/manifests/" + holder, + "/v2/web/config/blobs/sha256:def456", + } + if strings.Join(*asked, " ") != strings.Join(want, " ") { t.Fatalf("the store was asked %v; want %v", *asked, want) } } diff --git a/internal/broker/consumer_from_now_test.go b/internal/broker/consumer_from_now_test.go new file mode 100644 index 0000000..f56a607 --- /dev/null +++ b/internal/broker/consumer_from_now_test.go @@ -0,0 +1,91 @@ +package broker + +import ( + "fmt" + "os" + "testing" + "time" + + "github.com/nats-io/nats.go" +) + +// A consumer that reacts to announcements, made on a stream that keeps a week of them, starts from now: +// made from the start it replays every merge and build of that week (novox/hq issue 248). And a reset +// re-makes a stuck one from now, its configuration otherwise kept, and refuses a work queue. +// +// docker run -d --rm --name t -p 14231:4222 nats:2.10-alpine -js +// MESH_TEST_NATS=nats://127.0.0.1:14231 go test ./internal/broker/ -run TestAConsumerMadeFromNow +func TestAConsumerMadeFromNowNeverReplaysTheStreamsHistory(t *testing.T) { + url := os.Getenv("MESH_TEST_NATS") + if url == "" { + t.Skip("MESH_TEST_NATS unset") + } + js, err := Dial(url) + if err != nil { + t.Fatal(err) + } + defer js.Close() + const stream = "HISTORY_TEST" + _ = js.js.DeleteStream(stream) + if _, err := js.js.AddStream(&nats.StreamConfig{Name: stream, Subjects: []string{"history.>"}, Storage: nats.MemoryStorage}); err != nil { + t.Fatal(err) + } + defer func() { _ = js.js.DeleteStream(stream) }() + for i := 0; i < 5; i++ { + if _, err := js.js.Publish("history.merged", []byte(fmt.Sprint(i))); err != nil { + t.Fatal(err) + } + } + + fresh := Consumer{Name: "fresh", Stream: stream, Filters: []string{"history.merged"}, Push: true, + AckWaitSeconds: 30, MaxDeliver: 5, MaxAckPending: 1, FromNow: true} + if err := js.EnsureConsumer(fresh); err != nil { + t.Fatal(err) + } + info, _ := js.js.ConsumerInfo(stream, "fresh") + if info.NumPending != 0 || info.Config.DeliverPolicy != nats.DeliverNewPolicy { + t.Fatalf("a consumer made from now holds %d of the stream's past (policy %v)", info.NumPending, info.Config.DeliverPolicy) + } + _, _ = js.js.Publish("history.merged", []byte("new")) + time.Sleep(100 * time.Millisecond) + if info, _ = js.js.ConsumerInfo(stream, "fresh"); info.NumPending != 1 { + t.Fatalf("a new announcement is not pending: %d", info.NumPending) + } + // Asserted again, it keeps where it is. + if err := js.EnsureConsumer(fresh); err != nil { + t.Fatal(err) + } + + // One made from the start, as the server's default makes it, and stuck behind its history. + old := fresh + old.Name, old.FromNow = "old", false + if err := js.EnsureConsumer(old); err != nil { + t.Fatal(err) + } + if info, _ = js.js.ConsumerInfo(stream, "old"); info.NumPending != 6 { + t.Fatalf("the default should replay all six: %d", info.NumPending) + } + before, after, err := js.ResetConsumer(stream, "old") + if err != nil { + t.Fatal(err) + } + info, _ = js.js.ConsumerInfo(stream, "old") + if before.Pending != 6 || after.Pending != 0 || info.Config.MaxAckPending != 1 || info.Config.MaxDeliver != 5 || + info.Config.AckWait != 30*time.Second || info.Config.DeliverSubject == "" { + t.Fatalf("before %+v after %+v config %+v", before, after, info.Config) + } + + // Never a work queue: what is pending there is work. + const queue = "QUEUE_TEST" + _ = js.js.DeleteStream(queue) + if _, err := js.js.AddStream(&nats.StreamConfig{Name: queue, Subjects: []string{"queue.>"}, Retention: nats.WorkQueuePolicy, Storage: nats.MemoryStorage}); err != nil { + t.Fatal(err) + } + defer func() { _ = js.js.DeleteStream(queue) }() + if _, err := js.js.AddConsumer(queue, &nats.ConsumerConfig{Durable: "w", AckPolicy: nats.AckExplicitPolicy}); err != nil { + t.Fatal(err) + } + if _, _, err := js.ResetConsumer(queue, "w"); err == nil { + t.Fatal("a work queue's consumer was reset") + } +} diff --git a/internal/broker/derived.go b/internal/broker/derived.go index e83d662..6244c50 100644 --- a/internal/broker/derived.go +++ b/internal/broker/derived.go @@ -47,7 +47,13 @@ type Consumer struct { // and a merge that came back rebuilt what it had just built, five times over on 2026-09-30. // With one outstanding, the server holds the rest, and the heartbeat is keeping the message. MaxAckPending int - Why string + // FromNow makes a consumer that does not exist yet start at the stream's end rather than its + // beginning (novox/hq issue 248). For a consumer that reacts to announcements — a merge, a build's + // outcome — on a stream that keeps a week of them: the server's default, everything the stream + // holds, replays every merge and every build of that week as if it had just happened. A consumer + // that exists keeps where it is, whatever this says; only its making is decided here. + FromNow bool + Why string } // seatStreamName is the stream holding a seat's inbound work. Named after the seat rather than diff --git a/internal/broker/jetstream.go b/internal/broker/jetstream.go index 71a50e2..198ee68 100644 --- a/internal/broker/jetstream.go +++ b/internal/broker/jetstream.go @@ -308,6 +308,9 @@ func (j *JetStream) EnsureConsumer(c Consumer) error { } return nil case errors.Is(err, nats.ErrConsumerNotFound): + if c.FromNow { + want.DeliverPolicy = nats.DeliverNewPolicy + } if _, err := j.js.AddConsumer(c.Stream, want); err != nil { return fmt.Errorf("creating consumer %s on %s: %w", c.Name, c.Stream, err) } @@ -317,6 +320,51 @@ func (j *JetStream) EnsureConsumer(c Consumer) error { } } +// ConsumerState is what a reset says about a consumer, before and after. +type ConsumerState struct { + DeliverPolicy string + Delivered uint64 + AckFloor uint64 + Pending uint64 + AckPending int +} + +func stateOf(info *nats.ConsumerInfo) ConsumerState { + policy, _ := info.Config.DeliverPolicy.MarshalJSON() + return ConsumerState{DeliverPolicy: strings.Trim(string(policy), `"`), Delivered: info.Delivered.Stream, + AckFloor: info.AckFloor.Stream, Pending: info.NumPending, AckPending: info.NumAckPending} +} + +// ResetConsumer re-makes a consumer to start from now, its configuration otherwise unchanged (novox/hq +// issue 248): what it had not yet delivered or acknowledged is dropped, which is the point — on a stream +// that keeps history, a consumer replaying a week of announcements does nothing anyone wants. Refused on +// a work queue, where what is pending is work nobody else will do. +func (j *JetStream) ResetConsumer(stream, name string) (before, after ConsumerState, err error) { + info, err := j.js.StreamInfo(stream) + if err != nil { + return before, after, fmt.Errorf("asking about stream %s: %w", stream, err) + } + if info.Config.Retention == nats.WorkQueuePolicy { + return before, after, fmt.Errorf("%s is a work queue: what its consumer has pending is work, and a reset would drop it", stream) + } + have, err := j.js.ConsumerInfo(stream, name) + if err != nil { + return before, after, fmt.Errorf("asking about consumer %s on %s: %w", name, stream, err) + } + before = stateOf(have) + want := have.Config + want.DeliverPolicy = nats.DeliverNewPolicy + want.OptStartSeq, want.OptStartTime = 0, nil + if err := j.js.DeleteConsumer(stream, name); err != nil { + return before, after, fmt.Errorf("removing consumer %s on %s: %w", name, stream, err) + } + made, err := j.js.AddConsumer(stream, &want) + if err != nil { + return before, after, fmt.Errorf("re-making consumer %s on %s — it is gone until the controller asserts it at its next start: %w", name, stream, err) + } + return before, stateOf(made), nil +} + func retentionOf(r Retention) nats.RetentionPolicy { switch r { case RetentionWorkQueue: diff --git a/internal/broker/streams.go b/internal/broker/streams.go index 526cd65..87a32fa 100644 --- a/internal/broker/streams.go +++ b/internal/broker/streams.go @@ -270,6 +270,7 @@ func MeshConsumers() []Consumer { // announcement handed over behind it must wait on the server, not time out on the // client and come back to be acted on again. MaxAckPending: 1, + FromNow: true, Why: "the two events the mesh's own controller reacts to, one at a time; after " + "max-deliver it dead-letters, because an announcement it cannot act on will not " + "become actionable", diff --git a/internal/builder/registry.go b/internal/builder/registry.go index 49ff966..219b4fe 100644 --- a/internal/builder/registry.go +++ b/internal/builder/registry.go @@ -7,6 +7,8 @@ import ( "io" "net/http" "strings" + + "github.com/novox/mesh-controller/internal/artifacts" ) // Where built artifacts go. @@ -66,22 +68,32 @@ func (r Registry) PublishImage(ctx context.Context, localTag, repository string) return pinned, nil } -// PublishArchive stores bytes as a blob and returns where to fetch them from. +// PublishArchive stores bytes as a blob, holds it by a manifest, and returns where to fetch them +// from. // -// Two steps, which is the registry's own protocol: ask for somewhere to put it, then put it there -// naming the digest. The registry verifies the digest itself, so a blob that arrived corrupted is -// refused by the thing storing it rather than by the machine unpacking it a week later. +// Two steps for the blob, which is the registry's own protocol: ask for somewhere to put it, then +// put it there naming the digest. The registry verifies the digest itself, so a blob that arrived +// corrupted is refused by the thing storing it rather than by the machine unpacking it a week +// later. +// +// **Then a manifest that names it** (novox/hq issue 253, ADR 0189). The store's own collector +// marks only from manifests, and a blob no manifest names is collected however much the mesh +// means to keep it — so an archive published bare is an archive the first nightly collection +// deletes. The holder is composed from the digest and size alone (artifacts.Holder), which is +// what lets the sweep recompute it to backfill or let go without anything being recorded here. +// What a machine is told to fetch is the blob, exactly as before. func (r Registry) PublishArchive(ctx context.Context, repository string, body []byte, digest string) (string, error) { base := "http://" + r.Address + "/v2/" + repository final := base + "/blobs/" + digest // Already there. Blobs are immutable and named by their content, so this is not an // optimisation — re-uploading would be asking the registry to store what it already has under - // the name it already has. + // the name it already has. It is still held: a blob published before holders existed is + // exactly the one a rebuild of the same source finds already there. if there, err := r.has(ctx, final); err != nil { return "", err } else if there { - return final, nil + return final, r.hold(ctx, repository, digest, len(body)) } start, err := http.NewRequestWithContext(ctx, http.MethodPost, base+"/blobs/uploads/", nil) @@ -119,7 +131,17 @@ func (r Registry) PublishArchive(ctx context.Context, repository string, body [] said, _ := io.ReadAll(io.LimitReader(done.Body, 4096)) return "", fmt.Errorf("%s refused the blob: %s %s", base, done.Status, strings.TrimSpace(string(said))) } - return final, nil + return final, r.hold(ctx, repository, digest, len(body)) +} + +// hold puts the manifest holding an archive beside it. A build whose archive could not be held is +// a failed build: recorded as published, it would be an archive the store's collector takes. +func (r Registry) hold(ctx context.Context, repository, digest string, size int) error { + store := artifacts.Store{Address: r.Address, HTTP: r.client()} + if _, err := store.HoldBlob(ctx, repository, digest, int64(size)); err != nil { + return fmt.Errorf("published %s/blobs/%s and could not hold it by a manifest: %w", repository, digest, err) + } + return nil } // has is whether this registry already holds what is at that URL. diff --git a/internal/builder/registry_test.go b/internal/builder/registry_test.go index 20d4f00..1e3a440 100644 --- a/internal/builder/registry_test.go +++ b/internal/builder/registry_test.go @@ -9,6 +9,8 @@ import ( "net/http/httptest" "strings" "testing" + + "github.com/novox/mesh-controller/internal/artifacts" ) // An OCI registry as a content-addressed blob store, which is what it is. @@ -18,9 +20,10 @@ import ( // there is not sent again, and that a tag is never accepted as a pin. type fakeRegistry struct { - blobs map[string][]byte - uploads int - location string + blobs map[string][]byte + manifests map[string][]byte // "@" + puts map[string]int // blob uploads, by digest + location string } func (f *fakeRegistry) serve(t *testing.T) *httptest.Server { @@ -28,6 +31,8 @@ func (f *fakeRegistry) serve(t *testing.T) *httptest.Server { if f.blobs == nil { f.blobs = map[string][]byte{} } + f.manifests = map[string][]byte{} + f.puts = map[string]int{} mux := http.NewServeMux() server := httptest.NewServer(mux) mux.HandleFunc("/v2/", func(w http.ResponseWriter, r *http.Request) { @@ -39,8 +44,28 @@ func (f *fakeRegistry) serve(t *testing.T) *httptest.Server { return } w.WriteHeader(http.StatusNotFound) + case strings.Contains(r.URL.Path, "/manifests/"): + // The archive's holder (novox/hq issue 253): asked for with an Accept, put by digest. + repository, digest, _ := strings.Cut(strings.TrimPrefix(r.URL.Path, "/v2/"), "/manifests/") + switch r.Method { + case http.MethodHead: + if _, ok := f.manifests[repository+"@"+digest]; ok && + strings.Contains(r.Header.Get("Accept"), "application/vnd.oci.image.manifest.v1+json") { + w.WriteHeader(http.StatusOK) + return + } + w.WriteHeader(http.StatusNotFound) + case http.MethodPut: + body, _ := io.ReadAll(r.Body) + sum := sha256.Sum256(body) + if digest != "sha256:"+hex.EncodeToString(sum[:]) { + w.WriteHeader(http.StatusBadRequest) + return + } + f.manifests[repository+"@"+digest] = body + w.WriteHeader(http.StatusCreated) + } case r.Method == http.MethodPost && strings.HasSuffix(r.URL.Path, "/blobs/uploads/"): - f.uploads++ where := f.location if where == "" { where = "/v2/upload/" + hex.EncodeToString([]byte("session")) @@ -58,6 +83,7 @@ func (f *fakeRegistry) serve(t *testing.T) *httptest.Server { return } f.blobs[digest] = body + f.puts[digest]++ w.WriteHeader(http.StatusCreated) default: w.WriteHeader(http.StatusNotFound) @@ -92,6 +118,57 @@ func TestAnArchiveIsStoredAndFetchableByItsDigest(t *testing.T) { } } +func TestAnArchiveIsPublishedWithAManifestHoldingIt(t *testing.T) { + // The store's collector marks only from manifests, so a bare blob is one the first collection + // deletes, kept or not (novox/hq issue 253, ADR 0189). Every archive goes out held, by the + // holder the sweep can compute for itself from the digest and size. + f := &fakeRegistry{} + r := registryFor(t, f) + body := []byte("a theme") + sum := sha256.Sum256(body) + digest := "sha256:" + hex.EncodeToString(sum[:]) + + where, err := r.PublishArchive(context.Background(), "shell/config", body, digest) + if err != nil { + t.Fatal(err) + } + if !strings.HasSuffix(where, "/v2/shell/config/blobs/"+digest) { + t.Fatalf("a machine is told to fetch %q; the blob is still what is fetched", where) + } + manifest, holder := artifacts.Holder(digest, int64(len(body))) + if got := f.manifests["shell/config@"+holder]; string(got) != string(manifest) { + t.Fatalf("the store holds %v; want the holder %s in the archive's own repository", f.manifests, holder) + } + if !strings.Contains(string(manifest), `"digest":"`+digest+`"`) { + t.Fatalf("the holder does not name the archive: %s", manifest) + } + empty := sha256.Sum256([]byte("{}")) + if _, ok := f.blobs["sha256:"+hex.EncodeToString(empty[:])]; !ok { + t.Fatal("the holder's empty config was never put, and a registry refuses a manifest without it") + } +} + +func TestAnArchiveAlreadyThereIsStillHeld(t *testing.T) { + // A rebuild of the same source finds its archive already there — often one published bare, + // before holders. It is not sent again, and it is held. + f := &fakeRegistry{} + r := registryFor(t, f) + body := []byte("published bare") + sum := sha256.Sum256(body) + digest := "sha256:" + hex.EncodeToString(sum[:]) + f.blobs[digest] = body + + if _, err := r.PublishArchive(context.Background(), "shell/config", body, digest); err != nil { + t.Fatal(err) + } + if f.puts[digest] != 0 { + t.Fatal("a blob already there was sent again") + } + if _, holder := artifacts.Holder(digest, int64(len(body))); f.manifests["shell/config@"+holder] == nil { + t.Fatal("a blob already there was left bare") + } +} + func TestABlobAlreadyThereIsNotSentAgain(t *testing.T) { // Not an optimisation: blobs are named by their content, so re-uploading is asking the // registry to store what it already has under the name it already has. @@ -107,8 +184,11 @@ func TestABlobAlreadyThereIsNotSentAgain(t *testing.T) { if _, err := r.PublishArchive(context.Background(), "shell/config", body, digest); err != nil { t.Fatal(err) } - if f.uploads != 1 { - t.Fatalf("the blob was uploaded %d times", f.uploads) + if f.puts[digest] != 1 { + t.Fatalf("the blob was uploaded %d times", f.puts[digest]) + } + if len(f.manifests) != 1 { + t.Fatalf("publishing twice left %d holders; want the one", len(f.manifests)) } } diff --git a/internal/catalogue/verbs.go b/internal/catalogue/verbs.go index c52d527..97524af 100644 --- a/internal/catalogue/verbs.go +++ b/internal/catalogue/verbs.go @@ -96,6 +96,7 @@ var ControllerVerbs = []Verb{ Input: schema(map[string]string{ "id": "a plan's id (as `plans` lists them): that plan, tier by tier", "stop": "a plan's id: stop it — what was asked still builds, nothing further is asked", + "close": "a plan's id: close a plan that will not move again, as failed by hand (novox/hq issue 254)", "repository": "owner/repository: the plan a merge there would produce, saving nothing (what-if); with paths or modules", "paths": "with repository: the files the merge would change, comma-separated, from the repository's root", "modules": "with repository: or the modules it would change, comma-separated", diff --git a/internal/inventory/busrecords_test.go b/internal/inventory/busrecords_test.go index 15cf35d..fa9dfcb 100644 --- a/internal/inventory/busrecords_test.go +++ b/internal/inventory/busrecords_test.go @@ -42,26 +42,37 @@ func theSeatDeclarer() catalogue.Manifest { // A module assigned to a machine becomes a user with the authority its manifest declared — and the // protocol of a seat declared by a *different* module, which is the whole reason a seat exists. func TestAnAssignedModuleBecomesAUserWithWhatItDeclared(t *testing.T) { + // It declares where its account is delivered: a module with no own secret named broker can never + // be issued one, and is no user at all (novox/hq issue 195) — the case asserted below. shop := catalogue.Manifest{ Module: "shop", Version: "1", Emits: []string{"order.placed"}, Tools: []string{"price"}, - Uses: []string{"telegram-sender"}, + Uses: []string{"telegram-sender"}, + OwnSecrets: catalogue.OwnSecrets{"broker": {Path: "/run/broker"}}, } - inv, ctx := aMeshWith(t, theSeatDeclarer(), shop) + quiet := catalogue.Manifest{Module: "quiet", Version: "1", Emits: []string{"thing.happened"}} + inv, ctx := aMeshWith(t, theSeatDeclarer(), shop, quiet) if _, err := inv.AddNode(ctx, "one"); err != nil { t.Fatal(err) } - if _, err := inv.Assign(ctx, "one", "shop"); err != nil { - t.Fatal(err) + for _, module := range []string{"shop", "quiet"} { + if _, err := inv.Assign(ctx, "one", module); err != nil { + t.Fatal(err) + } } records, err := inv.BusRecords(ctx) if err != nil { t.Fatal(err) } - on := records.Assigned["one"] - if len(on) != 1 || on[0].Module != "shop" { - t.Fatalf("the machine's modules read as %+v", on) + var on []broker.Declared + for _, d := range records.Assigned["one"] { + if d.Module == "shop" { + on = append(on, d) + } + } + if len(on) != 1 || on[0].NoAccount { + t.Fatalf("the machine's modules read as %+v", records.Assigned["one"]) } if len(on[0].Uses) != 1 || on[0].Uses[0].Accepts[0] != "send" { t.Fatalf("the seat it uses carries no protocol: %+v — so it would be granted nothing on a "+ @@ -98,6 +109,11 @@ func TestAnAssignedModuleBecomesAUserWithWhatItDeclared(t *testing.T) { if !found { t.Fatal("no user was derived for the assigned module") } + for _, u := range users { + if u.Username() == "one.quiet" { + t.Fatal("a module with nowhere to read an account was made a user (novox/hq issue 195)") + } + } } // A machine holding a live token gets an enrolment user; one whose token is spent or expired does diff --git a/internal/inventory/collection.go b/internal/inventory/collection.go index 3f6cde4..684287a 100644 --- a/internal/inventory/collection.go +++ b/internal/inventory/collection.go @@ -3,6 +3,7 @@ package inventory import ( "context" "encoding/json" + "sort" "strings" "github.com/novox/mesh-controller/internal/catalogue" @@ -33,7 +34,8 @@ const KeptBuilds = 5 // - **a definition names it** — the reference appears in a module's recorded manifest, which is // what the mesh would hand a machine now. No age limit: this is the floor; // - **the mesh can still go back to it** — it is an artifact of one of the KeptBuilds most -// recent successful builds of its module; +// recent successful builds of a module the mesh still holds. A forgotten module keeps +// nothing beyond what a held definition names (novox/hq issue 253); // - it was already collected, in which case there is nothing left to do. // // Returned in a stated order so two runs over the same records ask for the same things in the @@ -109,12 +111,19 @@ func (i *Inventory) keptReferences(ctx context.Context) (map[string]bool, error) return nil, err } - // The KeptBuilds most recent successful builds of each module, whole. + // The KeptBuilds most recent successful builds of each module the mesh still holds, whole. + // + // **Only a module the mesh still holds can be gone back to** (novox/hq issue 253). "Somewhere + // to return to" is a reason about a module's releases; a module that has been forgotten has + // no releases left to return between, and its build rows stay only as history. Without the + // join every module ever built kept five builds' artifacts for ever — and once the store's + // collector runs for real, what the keep set says is what the disk holds. recent, err := i.store.Pool().Query(ctx, `select made from ( - select made, row_number() over (partition by module order by at desc, id desc) as back - from build - where failed = '' and module is not null and module <> '' + select b.made, row_number() over (partition by b.module order by b.at desc, b.id desc) as back + from build b + join module m on m.name = b.module + where b.failed = '' and b.module is not null and b.module <> '' ) ranked where back <= $1`, KeptBuilds) if err != nil { return nil, err @@ -169,6 +178,28 @@ func (i *Inventory) keptReferences(ctx context.Context) (map[string]bool, error) return keep, nil } +// KeptArchives is every archive the mesh keeps, in a stated order: the references the sweep must +// hold by a manifest before it lets anything go, and the ones an operator needs to read as all +// held before the store's collector is let loose (novox/hq issue 253, ADR 0189). +// +// Only references into the mesh's own store, and only blobs: an image is its own manifest, and a +// reference that is kept because nothing here can speak for it is not one the store can be asked +// about. +func (i *Inventory) KeptArchives(ctx context.Context) ([]string, error) { + keep, err := i.keptReferences(ctx) + if err != nil { + return nil, err + } + var out []string + for reference := range keep { + if path, ours := catalogue.InArtifactStore(reference); ours && strings.Contains(path, "/blobs/sha256:") { + out = append(out, reference) + } + } + sort.Strings(out) + return out, nil +} + // everyReferenceMade is every artifact reference any successful build recorded. func (i *Inventory) everyReferenceMade(ctx context.Context) ([]string, error) { rows, err := i.store.Pool().Query(ctx, diff --git a/internal/inventory/collection_test.go b/internal/inventory/collection_test.go index c7aaf77..e05a2a3 100644 --- a/internal/inventory/collection_test.go +++ b/internal/inventory/collection_test.go @@ -20,6 +20,19 @@ func ref(module, artifact string, n int) string { return fmt.Sprintf("%s%s/%s@sha256:%064x", catalogue.ArtifactStoreScheme, module, artifact, n) } +// holding registers a definition for each module that names no artifact, so the mesh holds the +// module and its recent builds are somewhere it can go back to — and nothing more. +func holding(t *testing.T, inv *Inventory, modules ...string) { + t.Helper() + for _, module := range modules { + m := catalogue.Manifest{Module: module, Version: "1"} + if err := inv.RegisterModule(context.Background(), m, + Source{Repository: "https://forge.invalid/" + module + ".git"}); err != nil { + t.Fatal(err) + } + } +} + // built records one successful build of a module publishing one image. func built(t *testing.T, inv *Inventory, id, module string, n int) string { t.Helper() @@ -35,6 +48,7 @@ func built(t *testing.T, inv *Inventory, id, module string, n int) string { func TestTheStoreKeepsTheRecentBuildsAndLetsGoOfTheRest(t *testing.T) { inv := fresh(t) ctx := context.Background() + holding(t, inv, "web") // Eight builds of one module, oldest first. Five are kept — the newest, and the four a // release that turns out wrong can be taken back to. @@ -98,6 +112,7 @@ func TestWhatHasBeenCollectedIsNotOfferedAgain(t *testing.T) { // time it runs, for ever — a number of requests that grows with the mesh's whole history. inv := fresh(t) ctx := context.Background() + holding(t, inv, "web") for i := 1; i <= 7; i++ { built(t, inv, fmt.Sprintf("b%02d", i), "web", i) } @@ -123,6 +138,7 @@ func TestWhatHasBeenCollectedIsNotOfferedAgain(t *testing.T) { func TestAFailedBuildNamesNothingToCollectAndEachModuleIsCountedOnItsOwn(t *testing.T) { inv := fresh(t) ctx := context.Background() + holding(t, inv, "web", "db") // A failed build published nothing, so it is neither kept nor collected — and it must not // count against the module's five. @@ -156,6 +172,7 @@ func TestAFailedBuildNamesNothingToCollectAndEachModuleIsCountedOnItsOwn(t *test func TestAnArtifactRecordedWithAnAddressIsOfferedAsTheMeshRecordsOne(t *testing.T) { inv := fresh(t) ctx := context.Background() + holding(t, inv, "tools") // The oldest build published the old way; five newer ones fill the module's five. old := aBuild("a00", "tools", "") @@ -190,3 +207,82 @@ func TestAnArtifactRecordedWithAnAddressIsOfferedAsTheMeshRecordsOne(t *testing. t.Fatalf("offered %v again after collecting it", again) } } + +// A forgotten module keeps nothing beyond what a held definition names (novox/hq issue 253). +// +// "Somewhere to go back to" is a reason about a module's releases, and a module the mesh no +// longer holds has none. Its build rows stay as history; its artifacts go — except one a module +// the mesh still holds names, which is the floor whatever built it. +func TestAForgottenModuleKeepsNothingAHeldDefinitionDoesNotName(t *testing.T) { + inv := fresh(t) + ctx := context.Background() + + // Three builds of a module that was never held, or was held and then forgotten: within its + // five, and kept for that reason until now. + var gone []string + for i := 1; i <= 3; i++ { + gone = append(gone, built(t, inv, fmt.Sprintf("o%02d", i), "old", 200+i)) + } + // A module the mesh holds, whose definition runs the forgotten module's newest image. + named := gone[2] + m := catalogue.Manifest{Module: "web", Version: "1", Resources: []map[string]any{{ + "id": "app", "type": "container", "name": "web", "image": named, + }}} + if err := inv.RegisterModule(ctx, m, Source{Repository: "https://forge.invalid/web.git"}); err != nil { + t.Fatal(err) + } + + go_, err := inv.ToCollect(ctx) + if err != nil { + t.Fatal(err) + } + if len(go_) != 2 || go_[0] != gone[0] || go_[1] != gone[1] { + t.Fatalf("offered %v; want %v — a forgotten module's builds are no release to go back to, "+ + "and only what a held definition names stays", go_, gone[:2]) + } + + // And once the module is held again, its five are kept again. + holding(t, inv, "old") + again, err := inv.ToCollect(ctx) + if err != nil { + t.Fatal(err) + } + if len(again) != 0 { + t.Fatalf("offered %v for a module the mesh holds, within its five", again) + } +} + +// The archives the mesh keeps are what the sweep holds before it lets anything go, and what an +// operator reads as all held before the store's collector is let loose (novox/hq issue 253). +func TestKeptArchivesAreTheKeptBlobsOnly(t *testing.T) { + inv := fresh(t) + ctx := context.Background() + holding(t, inv, "shell") + + archive := func(n int) string { + return fmt.Sprintf("%sshell/config/blobs/sha256:%064x", catalogue.ArtifactStoreScheme, n) + } + for i := 1; i <= 6; i++ { + b := aBuild(fmt.Sprintf("s%02d", i), "shell", "") + b.Made = []Artifact{ + {Name: "app", Kind: "image", Reference: ref("shell", "app", i)}, + {Name: "config", Kind: "archive", Reference: archive(i)}, + } + if err := inv.RecordBuild(ctx, b); err != nil { + t.Fatal(err) + } + } + kept, err := inv.KeptArchives(ctx) + if err != nil { + t.Fatal(err) + } + want := []string{archive(2), archive(3), archive(4), archive(5), archive(6)} + if len(kept) != len(want) { + t.Fatalf("kept archives %v; want the five recent ones and no images", kept) + } + for i := range want { + if kept[i] != want[i] { + t.Fatalf("kept archives %v; want %v", kept, want) + } + } +} diff --git a/internal/inventory/migrations/0058-a-newer-plan-supersedes-an-older.sql b/internal/inventory/migrations/0058-a-newer-plan-supersedes-an-older.sql new file mode 100644 index 0000000..bf3768f --- /dev/null +++ b/internal/inventory/migrations/0058-a-newer-plan-supersedes-an-older.sql @@ -0,0 +1,22 @@ +-- A newer plan supersedes the older open plans of the same repository and branch (novox/hq issue 254, +-- ADR 0218). +-- +-- A merge produced a plan without looking at the plans still open, so two merges a few minutes apart +-- were two plans working the same modules, and a plan stuck waiting on something that would never +-- come stayed open for ever beside the newer ones. The newer plan now takes over what the older had +-- not yet built and the older is closed as `superseded` — a state of its own, so `plans` can say +-- which plan replaced it rather than reading as a failure. +-- +-- `branch` is the branch the merge went into, so only a plan of the same branch is superseded. Empty +-- for every plan from before this was kept: which branch it answered is not known, and such a plan +-- is superseded by the next plan of its repository, whichever branch — nothing is lost by it, since +-- what it had not built is folded into the plan that supersedes it. +alter table release_plan add column branch text not null default ''; + +-- And the bus's user list the machine holding the bus was last sent, as a digest (novox/hq issue +-- 249). A module's new grants are refused by the bus until its user list says them, so that machine +-- is sent first whenever the list it would be sent differs from the one it was. Read from its whole +-- declaration, every pending change on it — a recorded upgrade the operator chose not to roll out — +-- went with every send anywhere. A digest and never the list (ADR 0043: the list is composed on each +-- push, never kept). Empty for a machine never sent one, which reads as behind once. +alter table node add column sent_bus_users text not null default ''; diff --git a/internal/inventory/nodes.go b/internal/inventory/nodes.go index fef70aa..e16e323 100644 --- a/internal/inventory/nodes.go +++ b/internal/inventory/nodes.go @@ -921,6 +921,26 @@ func (i *Inventory) RecordSent(ctx context.Context, node, digest string) error { return err } +// RecordSentBusUsers keeps a digest of the bus's user list a machine was just sent, by its name +// (novox/hq issue 249): whether the machine holding the bus must go first is whether this differs +// from the list composed now. +func (i *Inventory) RecordSentBusUsers(ctx context.Context, name, digest string) error { + _, err := i.store.Pool().Exec(ctx, + `update node set sent_bus_users = $2 where name = $1`, name, digest) + return err +} + +// SentBusUsers is the digest of the bus's user list a machine was last sent, empty for none. +func (i *Inventory) SentBusUsers(ctx context.Context, name string) (string, error) { + var sent string + err := i.store.Pool().QueryRow(ctx, + `select sent_bus_users from node where name = $1`, name).Scan(&sent) + if errors.Is(err, pgx.ErrNoRows) { + return "", nil + } + return sent, err +} + // Outstanding is the digest of the declaration a machine was last sent, by its name, and empty // for one that has never been sent anything. // diff --git a/internal/inventory/plans.go b/internal/inventory/plans.go index c68bafe..cf90380 100644 --- a/internal/inventory/plans.go +++ b/internal/inventory/plans.go @@ -15,16 +15,19 @@ import ( // the store so a controller replaced mid-plan resumes it, and so `status` can say what a merge // still waits for. type Plan struct { - ID string `json:"id"` - Repository string `json:"repository"` - Commit string `json:"commit"` - Created time.Time `json:"created"` - Updated time.Time `json:"updated"` - State string `json:"state"` - Tier int `json:"tier"` - Tiers [][]string `json:"tiers"` - Modules map[string]*PlanModule `json:"modules"` - Note string `json:"note,omitempty"` + ID string `json:"id"` + Repository string `json:"repository"` + // Branch is the branch the merge went into (novox/hq issue 254): a newer plan supersedes the open + // ones of the same repository and branch. Empty for a plan from before it was kept. + Branch string `json:"branch,omitempty"` + Commit string `json:"commit"` + Created time.Time `json:"created"` + Updated time.Time `json:"updated"` + State string `json:"state"` + Tier int `json:"tier"` + Tiers [][]string `json:"tiers"` + Modules map[string]*PlanModule `json:"modules"` + Note string `json:"note,omitempty"` } // PlanModule is one module's state within a plan. @@ -37,8 +40,15 @@ type PlanModule struct { // later tier is built by it (ADR 0163's gate): the reports that open the gate are the ones // after this. SentAt *time.Time `json:"sent_at,omitempty"` - Commit string `json:"commit,omitempty"` - Why string `json:"why,omitempty"` + // First is the machines the plan sent the new build to first, and FirstAt when (novox/hq issue + // 249, ADR 0218): unless the module's policy rolls it out together, one machine takes it before + // the rest, and the rest are sent once that one reports it applied. Kept so a controller + // replaced while the plan waits on that report resumes the wait rather than sending again. The + // machine holding the bus is among them when its user list had to go first. + First []string `json:"first,omitempty"` + FirstAt *time.Time `json:"first_at,omitempty"` + Commit string `json:"commit,omitempty"` + Why string `json:"why,omitempty"` } // The states a plan passes through. @@ -47,6 +57,9 @@ const ( PlanRolling = "rolling" PlanDone = "done" PlanFailed = "failed" + // PlanSuperseded is a plan a newer merge of the same repository and branch took over (novox/hq + // issue 254, ADR 0218): what it had not built is in the newer plan, and its note names it. + PlanSuperseded = "superseded" ) // Open says whether the plan is still being worked. @@ -63,11 +76,11 @@ func (i *Inventory) SavePlan(ctx context.Context, p Plan) error { return err } _, err = i.store.Pool().Exec(ctx, - `insert into release_plan (id, repository, commit_hash, created, updated, state, tier, tiers, modules, note) - values ($1, $2, $3, $4, now(), $5, $6, $7, $8, $9) + `insert into release_plan (id, repository, commit_hash, created, updated, state, tier, tiers, modules, note, branch) + values ($1, $2, $3, $4, now(), $5, $6, $7, $8, $9, $10) on conflict (id) do update set updated = now(), state = excluded.state, tier = excluded.tier, - tiers = excluded.tiers, modules = excluded.modules, note = excluded.note`, - p.ID, p.Repository, p.Commit, p.Created, p.State, p.Tier, tiers, modules, p.Note) + tiers = excluded.tiers, modules = excluded.modules, note = excluded.note, branch = excluded.branch`, + p.ID, p.Repository, p.Commit, p.Created, p.State, p.Tier, tiers, modules, p.Note, p.Branch) return err } @@ -95,7 +108,7 @@ func (i *Inventory) PlanByID(ctx context.Context, id string) (Plan, error) { func (i *Inventory) plans(ctx context.Context, tail string) ([]Plan, error) { rows, err := i.store.Pool().Query(ctx, - `select id, repository, commit_hash, created, updated, state, tier, tiers, modules, note + `select id, repository, commit_hash, created, updated, state, tier, tiers, modules, note, branch from release_plan `+tail) if err != nil { return nil, err @@ -106,7 +119,7 @@ func (i *Inventory) plans(ctx context.Context, tail string) ([]Plan, error) { var p Plan var tiers, modules []byte if err := rows.Scan(&p.ID, &p.Repository, &p.Commit, &p.Created, &p.Updated, &p.State, - &p.Tier, &tiers, &modules, &p.Note); err != nil { + &p.Tier, &tiers, &modules, &p.Note, &p.Branch); err != nil { return nil, err } if err := json.Unmarshal(tiers, &p.Tiers); err != nil { diff --git a/internal/inventory/plans_test.go b/internal/inventory/plans_test.go index a9153a6..e4804a7 100644 --- a/internal/inventory/plans_test.go +++ b/internal/inventory/plans_test.go @@ -46,3 +46,30 @@ func TestAPlanIsKeptAdvancedAndResumedFromTheStore(t *testing.T) { t.Fatalf("a done plan is still among the recent ones: %+v", recent) } } + +// novox/hq issue 254: a plan keeps the branch its merge went into, and a superseded plan is not open. +func TestASupersededPlanIsNotOpen(t *testing.T) { + inv := ForTest(t) + ctx := t.Context() + p := Plan{ID: "plan-1", Repository: "novox/mesh-catalog", Branch: "main", Commit: "abc", + Created: time.Now().UTC(), State: PlanBuilding, Tiers: [][]string{{"gitea"}}, + Modules: map[string]*PlanModule{"gitea": {}}} + if err := inv.SavePlan(ctx, p); err != nil { + t.Fatal(err) + } + kept, err := inv.PlanByID(ctx, "plan-1") + if err != nil || kept.Branch != "main" { + t.Fatalf("the branch was not kept: %v %+v", err, kept) + } + kept.State = PlanSuperseded + kept.Note = "superseded at tier 0 by plan-2" + if err := inv.SavePlan(ctx, kept); err != nil { + t.Fatal(err) + } + if open, err := inv.OpenPlans(ctx); err != nil || len(open) != 0 { + t.Fatalf("a superseded plan is still open: %v %+v", err, open) + } + if recent, _ := inv.RecentPlans(ctx, 5); len(recent) != 1 || recent[0].State != PlanSuperseded { + t.Fatalf("a superseded plan is not among the recent ones as superseded: %+v", recent) + } +} diff --git a/internal/inventory/sent_bus_users_test.go b/internal/inventory/sent_bus_users_test.go new file mode 100644 index 0000000..5717571 --- /dev/null +++ b/internal/inventory/sent_bus_users_test.go @@ -0,0 +1,25 @@ +package inventory + +import "testing" + +// novox/hq issue 249: the digest of the user list a machine was last sent is kept, by its name, and +// is empty for a machine never sent one. +func TestTheUserListAMachineWasSentIsKept(t *testing.T) { + inv := ForTest(t) + ctx := t.Context() + if _, err := inv.AddNode(ctx, "anchor"); err != nil { + t.Fatal(err) + } + if sent, err := inv.SentBusUsers(ctx, "anchor"); err != nil || sent != "" { + t.Fatalf("a machine never sent a list has %q: %v", sent, err) + } + if err := inv.RecordSentBusUsers(ctx, "anchor", "abc"); err != nil { + t.Fatal(err) + } + if sent, err := inv.SentBusUsers(ctx, "anchor"); err != nil || sent != "abc" { + t.Fatalf("the list sent was not kept: %q %v", sent, err) + } + if sent, err := inv.SentBusUsers(ctx, "nobody"); err != nil || sent != "" { + t.Fatalf("a machine the mesh does not know: %q %v", sent, err) + } +} diff --git a/internal/link/receive_nats_test.go b/internal/link/receive_nats_test.go index 6f2b779..ba276e0 100644 --- a/internal/link/receive_nats_test.go +++ b/internal/link/receive_nats_test.go @@ -315,6 +315,12 @@ func TestNatsTheEventsTheControllerFollowsArriveAndAreAcknowledged(t *testing.T) ctx, stop := context.WithCancel(context.Background()) defer stop() go func() { _ = s.Serve(ctx) }() + // The controller's event consumer is made from now (novox/hq issue 248): what was published before + // it existed is history it never replays. So the announcements are made once it is there. + eventually(t, "the controller's event consumer being made", func() bool { + _, err := js.Context().ConsumerInfo("EVENTS", broker.ControllerName) + return err == nil + }) moved, _ := json.Marshal(Upgraded{Module: "gitea", Commit: "abcdef0123"}) if _, err := js.Context().Publish(broker.ControllerFollows[0], moved); err != nil {