Files
mesh-controller/cmd/mesh-controller/plan_retry.go
T
jschoubben 83ce19b9c0
mesh/merge-gate pass: builds build-agent, mesh-controller → ace, g14, novox, shanks; no bus step; every machine composes with the change as it did without …
mesh/repo-check pass: its merge-check.sh passed
mesh/delivery delivered
Say a registry came back only when the call worked, and retry from toRetry's set (issue 457 review)
The streaming blob PUT is left unwaited, with why, since its body cannot be
read twice and the POST before it already waited.
2026-10-11 11:05:30 +02:00

551 lines
20 KiB
Go

package main
import (
"context"
"fmt"
"os"
"slices"
"sort"
"strings"
"time"
"github.com/novox/mesh-controller/internal/inventory"
"github.com/novox/mesh-controller/internal/link"
)
// A plan follows what is done to the build queue (novox/hq ADR 0219).
//
// Two things a person does to builds by hand would otherwise leave a plan saying something untrue:
//
// - **Pausing the build seat.** A plan whose builds wait in the queue of a seat whose every holder
// is paused is not late — nothing will take its asks until somebody resumes — and saying LATE
// sends a reader looking for a fault that is a decision. It says what it waits on instead.
// - **A build that failed, been cancelled or killed, and is asked again.** `plans retry` re-asks a
// failed plan's failed modules and the plan goes on from that tier as if they had built the
// first time; `rebuild` of a module a plan holds unbuilt joins that plan rather than running
// beside it — beside it, the plan would either ask it again or stay failed on an outcome the
// rebuild has already replaced.
// pauseView is whether the build seat takes work: the holders that said they are paused, and
// whether that is every holder.
type pauseView struct {
Nodes []string
All bool
}
// pausedWaiting is what a plan waits on when the build seat is paused under it, or false: a plan
// building, whose current tier has an ask outstanding, while every holder of the seat is paused.
func pausedWaiting(p inventory.Plan, pause pauseView, now time.Time) (string, bool) {
if p.State != inventory.PlanBuilding || !pause.All || len(pause.Nodes) == 0 || p.Tier >= len(p.Tiers) {
return "", false
}
var asked *time.Time
for _, m := range p.Tiers[p.Tier] {
if s := p.Modules[m]; s != nil && s.State == "asked" && s.AskedAt != nil {
if asked == nil || s.AskedAt.Before(*asked) {
asked = s.AskedAt
}
}
}
if asked == nil {
return "", false
}
waited := now.Sub(*asked)
said := fmt.Sprintf("waiting: the build seat is paused on %s (asked %s ago)",
strings.Join(pause.Nodes, ", "), waited.Round(time.Second))
// Not late — a pause is a decision — but a pause forgotten is a plan that never moves, so one held
// longer than a day is named.
if waited > pausedTooLong {
said += " — PAUSED OVER A DAY: `resume` takes builds again"
}
return said, true
}
// pausedTooLong is how long a plan may wait on a paused build seat before it says the pause is long.
const pausedTooLong = 24 * time.Hour
// awaitsABuild is whether any open plan has an ask outstanding in its current tier — the only case
// where the seat being paused changes what a plan says.
func awaitsABuild(plans []inventory.Plan) bool {
for _, p := range plans {
if p.State != inventory.PlanBuilding || p.Tier >= len(p.Tiers) {
continue
}
for _, m := range p.Tiers[p.Tier] {
if s := p.Modules[m]; s != nil && s.State == "asked" {
return true
}
}
}
return false
}
// buildSeatPause reads whether the build seat's holders take work, from what each last said on the
// bus (link.HolderState) — read only when a plan waits on a build, so a mesh with nothing building
// does not dial the bus to say so. Anything unreadable is said and read as not paused: a plan then
// reads as late, which is what it said before this existed.
func buildSeatPause(ctx context.Context, inv *inventory.Inventory, plans []inventory.Plan) pauseView {
if !awaitsABuild(plans) {
return pauseView{}
}
entries, err := inv.Catalogued(ctx)
if err != nil {
return pauseView{}
}
seat := buildSeatAmong(entries)
holders := holdersAmong(entries, seat)
if len(holders) == 0 {
return pauseView{}
}
// On the serving controller's own connection when this is it: a watchdog tick while a walk waits for a
// build dialled one every 30 seconds (novox/hq issue 327).
js, err := aBus()
if err != nil {
fmt.Fprintf(os.Stderr, "could not reach the bus to read whether the build seat is paused: %v\n", err)
return pauseView{}
}
defer js.Close()
said, err := link.PausedSaid(js, seat, holders)
if err != nil {
fmt.Fprintf(os.Stderr, "could not read whether the build seat is paused: %v\n", err)
return pauseView{}
}
return pauseOf(holders, said)
}
// pauseOf is the view from what each holder said.
func pauseOf(holders []string, said map[string]link.HolderState) pauseView {
var v pauseView
for _, n := range holders {
if said[n].Paused {
v.Nodes = append(v.Nodes, n)
}
}
sort.Strings(v.Nodes)
v.All = len(holders) > 0 && len(v.Nodes) == len(holders)
return v
}
// failedIn is the modules of a plan's current tier that failed to build, sorted.
func failedIn(p inventory.Plan) []string {
if p.Tier >= len(p.Tiers) {
return nil
}
var out []string
for _, m := range p.Tiers[p.Tier] {
if s := p.Modules[m]; s != nil && s.State == "failed" {
out = append(out, m)
}
}
sort.Strings(out)
return out
}
// newerOpenPlan is an open plan of the same repository and branch made after this one: the plan
// that holds what this one held now (issue 254, ADR 0218).
func newerOpenPlan(p inventory.Plan, plans []inventory.Plan) (inventory.Plan, bool) {
for _, q := range plans {
if q.ID == p.ID || !q.Open() || !strings.EqualFold(q.Repository, p.Repository) ||
(p.Branch != "" && q.Branch != "" && p.Branch != q.Branch) || !q.Created.After(p.Created) {
continue
}
return q, true
}
return inventory.Plan{}, false
}
// retryRefusal is why a plan cannot be retried, or nothing.
func retryRefusal(p inventory.Plan, plans []inventory.Plan) error {
switch {
case p.State == inventory.PlanDone:
return fmt.Errorf("%s is done; there is nothing to retry", p.ID)
case p.State == inventory.PlanSuperseded:
return fmt.Errorf("%s was %s — what it had not built is in that plan", p.ID, p.Note)
case p.Open():
return fmt.Errorf("%s is still %s; nothing in it failed to retry — `rebuild <module>` asks one module again", p.ID, p.State)
}
if len(failedIn(p)) > 0 {
if q, found := newerOpenPlan(p, plans); found {
return fmt.Errorf("%s supersedes it: a newer merge of %s (%s at %s) is open, and retrying %s would build "+
"what that one replaced", q.ID, q.Repository, q.ID, short(q.Commit), p.ID)
}
return oneWalkAtATime(p, plans)
}
stopped := stoppedRollouts(p)
if len(stopped) == 0 {
return fmt.Errorf("nothing in tier %d of %s failed to build or stopped rolling out — it stopped at: %s",
p.Tier, p.ID, p.Note)
}
// **A build that failed its gate is not sent again** (novox/hq ADR 0236): it was put back on its first
// machine, and retrying would judge the build the mesh put back, or send the failed one by hand.
for _, m := range stopped {
if g := p.Modules[m].Gate; g != nil && g.Verdict == inventory.GateFailed && !noVerdictOnItsBuild(g) {
return fmt.Errorf("%s failed its gate on %s (%s) and was put back: a build that failed its gate is not "+
"sent again — a newer merge, or `rebuild %s`, makes a new build, judged at the gate again",
m, strings.Join(g.Machines, ", "), g.Why, m)
}
}
// **A rollout is retried unless the module has moved on**: a newer plan holding it sends — or
// sent — a newer build, and sending this one again would put the older build back on its machines.
for _, m := range stopped {
if q, found := newerPlanFor(m, p, plans); found {
return fmt.Errorf("%s has a newer plan, %s (%s, %s at %s): sending %s's build of it again would put "+
"the older build back", m, q.ID, q.State, q.Repository, short(q.Commit), p.ID)
}
}
return oneWalkAtATime(p, plans)
}
// oneWalkAtATime refuses a retry while another walk is open, started or waiting for its word (novox/hq ADR
// 0276): a walk retried beside it would be two walks at once.
func oneWalkAtATime(p inventory.Plan, plans []inventory.Plan) error {
for _, q := range plans {
if q.ID != p.ID && q.Open() && q.Release == nil {
return fmt.Errorf("%s is open (%s): one walk at a time — retry %s once it ended", q.ID, q.Named(), p.ID)
}
}
return nil
}
// stoppedRollouts is the modules of a plan's current tier whose rollout stopped at its first machine
// (issue 249, ADR 0218): built, sent to the first machine, never to the rest, and why it stopped kept.
func stoppedRollouts(p inventory.Plan) []string {
if p.Tier >= len(p.Tiers) {
return nil
}
var out []string
for _, m := range p.Tiers[p.Tier] {
if s := p.Modules[m]; s != nil && s.State == "built" && s.FirstAt != nil && s.SentAt == nil &&
len(s.First) > 0 && s.Why != "" {
out = append(out, m)
}
}
sort.Strings(out)
return out
}
// newerPlanFor is a plan made after this one that holds the module, superseded ones aside.
func newerPlanFor(module string, p inventory.Plan, plans []inventory.Plan) (inventory.Plan, bool) {
for _, q := range plans {
if q.ID == p.ID || q.State == inventory.PlanSuperseded || !q.Created.After(p.Created) {
continue
}
for _, tier := range q.Tiers {
for _, m := range tier {
if m == module {
return q, true
}
}
}
}
return inventory.Plan{}, false
}
// sendRollout sends machines what the mesh would send them now, answering the ones it sent. A
// variable so a test of a retried rollout needs no machine.
var sendRollout = sendToEach
// resumed sets a failed plan building again once nothing in its tier is failed.
func resumed(p *inventory.Plan, why string) {
if p.State == inventory.PlanFailed && len(failedIn(*p)) == 0 {
p.State = inventory.PlanBuilding
p.Note = why
}
}
// retryPlan asks a failed plan's failed modules again, under new ids, and sets it building again at
// that tier: what follows is the plan going on as if they had built the first time.
func retryPlan(ctx context.Context, open *stores, id string) (string, error) {
inv := open.inventory
release, err := inv.HoldPlans(ctx, true)
if err != nil {
return "", err
}
defer release()
p, err := inv.PlanByID(ctx, id)
if err != nil {
return "", err
}
plans, err := inv.OpenPlans(ctx)
if err != nil {
return "", err
}
recent, err := inv.RecentPlans(ctx, 50)
if err != nil {
return "", err
}
plans = append(plans, recent...)
// **A failed walk whose earlier merges are walked alone is not retried** (novox/hq ADR 0276): the search for
// the merge that brought the failure answers them now, and a retried walk would name them twice.
if p.Delivery != nil && len(p.Delivery.Merges) > 0 {
kept, err := inv.MergesOf(ctx, p.ID)
if err != nil {
return "", err
}
if len(kept) < len(p.Delivery.Merges) {
return "", fmt.Errorf("%s's earlier merges are walked alone, to find which one brought its failure: "+
"those walks answer them; a newer merge, or `rebuild <module>`, builds again", p.ID)
}
}
// Settled from the build records before anything is judged (novox/hq issue 457): a build of the tier
// that failed after the plan did is as failed as the one that failed it. What toRetry says is the
// failed set every step below works from.
var failed []string
if p.Tier < len(p.Tiers) {
recorded, byID, err := recordsOfAsked(ctx, inv, &p, p.Tiers[p.Tier])
if err != nil {
return "", err
}
failed = toRetry(&p, recorded, byID)
}
if err := retryRefusal(p, plans); err != nil {
return "", err
}
if again := unjudgedAtGate(p); len(failed) == 0 && len(again) > 0 {
return retryTierWhole(ctx, open, &p, again)
}
if len(failed) == 0 {
return retryRollouts(ctx, open, &p)
}
entries, err := inv.Catalogued(ctx)
if err != nil {
return "", err
}
byName := map[string]inventory.Entry{}
for _, e := range entries {
byName[e.Manifest.Module] = e
}
var asked []string
for _, m := range failed {
askModule(ctx, &p, m, byName)
if s := p.Modules[m]; s.State == "asked" {
asked = append(asked, m+" as "+s.Build)
}
}
resumed(&p, fmt.Sprintf("tier %d retried by hand: %s asked again", p.Tier, strings.Join(failed, ", ")))
if err := inv.SavePlan(ctx, &p); err != nil {
return "", err
}
if p.State != inventory.PlanBuilding {
return "", fmt.Errorf("%s could not be resumed: %s", p.ID, p.Note)
}
return fmt.Sprintf("%s retried at tier %d of %d: asked %s; the plan goes on from there as any plan does",
p.ID, p.Tier, len(p.Tiers), strings.Join(asked, ", ")), nil
}
// joinAPlan asks a module again for the plan that holds it unbuilt or failed, if one does: open plans
// first, then the most recent failed one that nothing newer supersedes. Says whether it joined one.
func joinAPlan(ctx context.Context, open *stores, module string) (bool, string, error) {
inv := open.inventory
release, err := inv.HoldPlans(ctx, true)
if err != nil {
return false, "", err
}
defer release()
openPlans, err := inv.OpenPlans(ctx)
if err != nil {
return false, "", err
}
recent, err := inv.RecentPlans(ctx, 20)
if err != nil {
return false, "", err
}
p, found := planHolding(module, openPlans, recent)
if !found {
return false, "", nil
}
entries, err := inv.Catalogued(ctx)
if err != nil {
return false, "", err
}
byName := map[string]inventory.Entry{}
for _, e := range entries {
byName[e.Manifest.Module] = e
}
was := p.State
askModule(ctx, &p, module, byName)
s := p.Modules[module]
resumed(&p, fmt.Sprintf("tier %d: %s rebuilt by hand", p.Tier, module))
if err := inv.SavePlan(ctx, &p); err != nil {
return false, "", err
}
if s.State != "asked" {
return true, "", fmt.Errorf("%s could not be asked for %s: %s", module, p.ID, s.Why)
}
said := fmt.Sprintf("rebuild asked as %s, joining %s at tier %d (%s at %s): the plan takes this build as %s's outcome",
s.Build, p.ID, p.Tier, p.Repository, short(p.Commit), module)
switch {
case was == inventory.PlanFailed && p.State == inventory.PlanBuilding:
said += "; the plan had failed and builds again from this tier"
case was == inventory.PlanFailed:
said += "; the plan stays failed while " + strings.Join(failedIn(p), ", ") + " failed too — `plans retry " + p.ID + "` asks them"
}
return true, said, nil
}
// planHolding is the plan a rebuild of a module joins: an open plan whose current tier holds it not
// yet built, else the newest failed plan whose current tier does, and that no open plan of its
// repository supersedes.
func planHolding(module string, openPlans, recent []inventory.Plan) (inventory.Plan, bool) {
holds := func(p inventory.Plan) bool {
if p.Tier >= len(p.Tiers) {
return false
}
for _, m := range p.Tiers[p.Tier] {
if m == module {
s := p.Modules[m]
return s == nil || s.State != "built"
}
}
return false
}
for _, p := range openPlans {
if holds(p) {
return p, true
}
}
sorted := append([]inventory.Plan(nil), recent...)
sort.SliceStable(sorted, func(i, j int) bool { return sorted[i].Created.After(sorted[j].Created) })
for _, p := range sorted {
if p.State != inventory.PlanFailed || !holds(p) {
continue
}
if _, superseded := newerOpenPlan(p, openPlans); superseded {
continue
}
return p, true
}
return inventory.Plan{}, false
}
// retryRollouts sends each module whose rollout stopped to the machines it was first sent to, again,
// records that send as the first anew, and sets the plan rolling: from there it goes on as the plan
// would have — the rest sent once those report they applied it, the next tier after (ADR 0218).
func retryRollouts(ctx context.Context, open *stores, p *inventory.Plan) (string, error) {
// Every machine once, for all the modules stopped there (novox/hq issue 281).
stopped := stoppedRollouts(*p)
var machines []string
for _, m := range stopped {
for _, n := range p.Modules[m].First {
if !slices.Contains(machines, n) {
machines = append(machines, n)
}
}
}
sort.Strings(machines)
sent, err := sendRollout(ctx, open, machines)
if err != nil {
return "", fmt.Errorf("%s could not be sent to %s again, so %s stays failed: %w",
strings.Join(stopped, ", "), strings.Join(machines, ", "), p.ID, err)
}
now := time.Now().UTC()
var said []string
for _, m := range stopped {
s := p.Modules[m]
var again []string
for _, n := range sent {
if slices.Contains(s.First, n) {
again = append(again, n)
}
}
s.First, s.FirstAt, s.Why = again, &now, ""
said = append(said, m+" to "+strings.Join(again, ", "))
}
p.State = inventory.PlanRolling
p.Note = fmt.Sprintf("tier %d retried by hand; sent %s first again", p.Tier, strings.Join(said, "; "))
if err := open.inventory.SavePlan(ctx, p); err != nil {
return "", err
}
return fmt.Sprintf("%s retried at tier %d of %d: sent %s first again; the rest follow once it reports it "+
"applied, as the plan would have", p.ID, p.Tier, len(p.Tiers), strings.Join(said, "; ")), nil
}
// noVerdictOnItsBuild says a failed gate said nothing about the module's build (novox/hq issue 281): its
// send changed nothing of the module there — the machine already ran that build, carried there by an
// earlier send of the same tier, or one identical to it. Such a module was blamed for its machine.
func noVerdictOnItsBuild(g *inventory.PlanGate) bool {
return g.Rollback == gateUnchanged || (g.From != "" && sameCommit(g.From, g.To))
}
// unjudgedAtGate is the modules of a plan's current tier stopped at a gate that was no verdict on their
// build, sorted.
func unjudgedAtGate(p inventory.Plan) []string {
if p.Tier >= len(p.Tiers) {
return nil
}
var out []string
for _, m := range p.Tiers[p.Tier] {
if s := p.Modules[m]; s != nil && s.Gate != nil && s.Gate.Verdict == inventory.GateFailed && noVerdictOnItsBuild(s.Gate) {
out = append(out, m)
}
}
sort.Strings(out)
return out
}
// retryTierWhole retries a plan stopped at a gate that judged no build of the module it stopped on
// (issue 281): that module is asked again under a new id — its old build may be marked failed, and a new
// verdict is what takes the gate's condition away — and every module of the tier sent first and never
// passed is sent again, the tier whole, one send per machine, judged again. What passed stays passed.
func retryTierWhole(ctx context.Context, open *stores, p *inventory.Plan, again []string) (string, error) {
inv := open.inventory
entries, err := inv.Catalogued(ctx)
if err != nil {
return "", err
}
byName := map[string]inventory.Entry{}
for _, e := range entries {
byName[e.Manifest.Module] = e
}
var resent []string
for _, m := range p.Tiers[p.Tier] {
s := p.Modules[m]
if s == nil || s.FirstAt == nil || s.SentAt != nil || slices.Contains(again, m) ||
(s.Gate != nil && s.Gate.Verdict == inventory.GatePassed) {
continue
}
sendAgain(s)
resent = append(resent, m)
}
var asked []string
for _, m := range again {
sendAgain(p.Modules[m])
askModule(ctx, p, m, byName)
if s := p.Modules[m]; s.State == "asked" {
asked = append(asked, m+" as "+s.Build)
}
}
if p.State == inventory.PlanFailed && len(failedIn(*p)) == 0 {
p.State = inventory.PlanBuilding
p.Note = fmt.Sprintf("tier %d retried by hand: %s asked again; %s sent again with the tier", p.Tier,
strings.Join(again, ", "), orNone(strings.Join(resent, ", ")))
}
if err := inv.SavePlan(ctx, p); err != nil {
return "", err
}
if p.State != inventory.PlanBuilding {
return "", fmt.Errorf("%s could not be resumed: %s", p.ID, p.Note)
}
return fmt.Sprintf("%s retried at tier %d of %d: asked %s; %s sent again with the tier, one send per machine, "+
"judged again", p.ID, p.Tier, len(p.Tiers), strings.Join(asked, ", "), orNone(strings.Join(resent, ", "))), nil
}
// sendAgain forgets a module's first send, so its plan sends it again.
func sendAgain(s *inventory.PlanModule) {
s.First, s.FirstAt, s.Gate, s.GatedBy, s.Previous, s.Why = nil, nil, nil, "", "", ""
}
// toRetry is the modules a retry asks again: every one of the tier that failed, the plan's state
// settled from the build records first (novox/hq issue 457). A plan fails on the first failure in its
// tier, and an outcome arriving after that finds no open plan to answer — it is kept only in the build
// records. Read from the plan alone, a retry asked only the build that failed first, and the records
// then failed the plan again on the next: each failed build of a tier took a retry of its own. What
// still runs is left asked, and its outcome is the plan's once the retry sets it building.
func toRetry(p *inventory.Plan, recorded map[string][]inventory.Build, byID map[string]inventory.Build) []string {
if p.Tier < len(p.Tiers) {
settleFromRecords(p, p.Tiers[p.Tier], recorded, byID)
}
return failedIn(*p)
}