diff --git a/cmd/mesh-controller/plan_retry.go b/cmd/mesh-controller/plan_retry.go index 10ed9c45..df28926e 100644 --- a/cmd/mesh-controller/plan_retry.go +++ b/cmd/mesh-controller/plan_retry.go @@ -289,21 +289,23 @@ func retryPlan(ctx context.Context, open *stores, id string) (string, error) { } } // Settled from the build records before anything is judged (novox/hq issue 457): a build of the tier - // that failed after the plan did is as failed as the one that failed it. + // that failed after the plan did is as failed as the one that failed it. What toRetry says is the + // failed set every step below works from. + var failed []string if p.Tier < len(p.Tiers) { recorded, byID, err := recordsOfAsked(ctx, inv, &p, p.Tiers[p.Tier]) if err != nil { return "", err } - toRetry(&p, recorded, byID) + failed = toRetry(&p, recorded, byID) } if err := retryRefusal(p, plans); err != nil { return "", err } - if again := unjudgedAtGate(p); len(failedIn(p)) == 0 && len(again) > 0 { + if again := unjudgedAtGate(p); len(failed) == 0 && len(again) > 0 { return retryTierWhole(ctx, open, &p, again) } - if len(failedIn(p)) == 0 { + if len(failed) == 0 { return retryRollouts(ctx, open, &p) } entries, err := inv.Catalogued(ctx) @@ -314,7 +316,6 @@ func retryPlan(ctx context.Context, open *stores, id string) (string, error) { for _, e := range entries { byName[e.Manifest.Module] = e } - failed := failedIn(p) var asked []string for _, m := range failed { askModule(ctx, &p, m, byName) diff --git a/internal/builder/mirror.go b/internal/builder/mirror.go index 8840e3d5..4a2da87e 100644 --- a/internal/builder/mirror.go +++ b/internal/builder/mirror.go @@ -429,6 +429,11 @@ func (r Registry) copyBlob(ctx context.Context, src *source, where upstream, dig if response.ContentLength > 0 { put.ContentLength = response.ContentLength } + // **Not waited for if refused** (novox/hq issue 457): the body streams from upstream and cannot be + // read twice, so a registry held still between the POST above and this PUT fails the copy with + // "its body cannot be read twice" rather than waiting. Accepted: the POST a moment before already + // waited the registry out, so the window is the length of one upstream fetch, and the build fails + // loudly, to be asked again, rather than buffering every base blob in memory. done, err := r.client().Do(put) if err != nil { return fmt.Errorf("cannot upload blob %s: %w", digest, err) diff --git a/internal/builder/registry_wait.go b/internal/builder/registry_wait.go index 010622a0..f2b10bae 100644 --- a/internal/builder/registry_wait.go +++ b/internal/builder/registry_wait.go @@ -74,8 +74,14 @@ func waitForRegistry(ctx context.Context, address, what string, try func() error } pause = min(pause*2, registryMostPause) if err = try(); !refused(err) { - tell("registry", "%s: the registry at %s answers again after %s; the build goes on", what, address, - time.Since(started).Round(time.Millisecond)) + // Said as it is: the registry answering is only the build going on when the call worked. + if err == nil { + tell("registry", "%s: the registry at %s answers again after %s; the build goes on", what, address, + time.Since(started).Round(time.Millisecond)) + } else { + tell("registry", "%s: the registry at %s no longer refuses after %s, and answered with: %v", what, + address, time.Since(started).Round(time.Millisecond), err) + } return err } }