From 5d1f7762c2bc4daa122025adcb5c9657beec5e2b Mon Sep 17 00:00:00 2001 From: jochen Date: Mon, 14 Sep 2026 19:58:01 +0200 Subject: [PATCH] One archive per machine, not one per image MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The images a bed stocks are almost entirely the same bytes: the base they share is 227 MB and a module's own code is a few. Exported one at a time that base is written, pushed and loaded once per image — for this bed, the same 227 MB crossed thirty-odd times, and the whole set measured 9.7 GB. Measured on eight of them: 2.24 GB as separate archives, 0.29 GB as one. 87 per cent less, and it improves with the count. This is the slowest thing a raise does, and the four-machine bed has been exceeding its own ninety-minute limit while still copying — so it was failing on the clock rather than on anything it was testing. Also stops shipping a compiler in every module image. The image runs compiled code and never compiles any; tsc runs on the workstation. Worth 26 MB an image, which is small beside the above but was pure waste. --- scripts/build-module-runtime.sh | 8 ++ scripts/build-runtime-image.sh | 10 ++- src/lifecycle/place.ts | 140 +++++++++++++++++++------------- 3 files changed, 99 insertions(+), 59 deletions(-) diff --git a/scripts/build-module-runtime.sh b/scripts/build-module-runtime.sh index cff3e8f..721cd83 100755 --- a/scripts/build-module-runtime.sh +++ b/scripts/build-module-runtime.sh @@ -48,6 +48,14 @@ rm -rf "$STAGE/node_modules/@novox/mesh-sdk" mkdir -p "$STAGE/node_modules/@novox" cp -rL "$MESH_SDK" "$STAGE/node_modules/@novox/mesh-sdk" rm -rf "$STAGE/node_modules/@novox/mesh-sdk/node_modules" +# **The image runs compiled code and never compiles any**, so it does not need a compiler. The +# tree copied above is the runtime's full install, development dependencies and all — and the +# compiler alone is 23 of its 28 MB. Every module image carried one, on every machine, for nothing: +# tsc runs on the workstation a few lines above, not in here. +# +# Removed by name rather than by `npm prune --omit=dev`, which would re-resolve dependencies — one +# of them a git URL with no registry behind it — and could drop something the image needs. +rm -rf "$STAGE/node_modules/typescript" "$STAGE/node_modules/@types" mkdir -p "$STAGE/modules/$MODULE"; cp -r "$MOD/dist" "$STAGE/modules/$MODULE/dist" cp "$MESH_TOOLS/package.json" "$STAGE/package.json" diff --git a/scripts/build-runtime-image.sh b/scripts/build-runtime-image.sh index e6ca133..5fce11d 100755 --- a/scripts/build-runtime-image.sh +++ b/scripts/build-runtime-image.sh @@ -46,7 +46,15 @@ cp -rL "$MESH_TOOLS/node_modules" "$STAGE/node_modules" rm -rf "$STAGE/node_modules/@novox/mesh-sdk" mkdir -p "$STAGE/node_modules/@novox" cp -rL "$MESH_SDK" "$STAGE/node_modules/@novox/mesh-sdk" -rm -rf "$STAGE/node_modules/@novox/mesh-sdk/node_modules" # -L materialises the @novox/mesh-sdk symlink +rm -rf "$STAGE/node_modules/@novox/mesh-sdk/node_modules" +# **The image runs compiled code and never compiles any**, so it does not need a compiler. The +# tree copied above is the runtime's full install, development dependencies and all — and the +# compiler alone is 23 of its 28 MB. Every module image carried one, on every machine, for nothing: +# tsc runs on the workstation a few lines above, not in here. +# +# Removed by name rather than by `npm prune --omit=dev`, which would re-resolve dependencies — one +# of them a git URL with no registry behind it — and could drop something the image needs. +rm -rf "$STAGE/node_modules/typescript" "$STAGE/node_modules/@types" # -L materialises the @novox/mesh-sdk symlink mkdir -p "$STAGE/modules/audit-logger" cp -r "$AUDIT/dist" "$STAGE/modules/audit-logger/dist" cp "$MESH_TOOLS/package.json" "$STAGE/package.json" diff --git a/src/lifecycle/place.ts b/src/lifecycle/place.ts index be85aec..8f2a785 100644 --- a/src/lifecycle/place.ts +++ b/src/lifecycle/place.ts @@ -498,20 +498,16 @@ export async function loadHeldImages( if (plans.length === 0) return []; const held: HeldImage[] = []; + + // Validate everything first, so a scenario naming something it should not is refused before any + // machine is touched rather than halfway through a transfer. for (const requested of scenario.images ?? []) { - // Refused by the validator, so reaching here would be a validator bug — but the consequence - // is a third-party image quietly loaded from the workstation instead of pulled, which is the - // fiction all of this exists to remove. Cheap to check, expensive to miss. if (!mustBeHandedOver(requested)) { throw new Error( `images: '${requested}' is not one of the mesh's own images. It is pulled from the ` + `internet by the machine that needs it, not loaded from this workstation.`, ); } - - const wanted = plans.filter((plan) => plan.images.includes(requested)).map((p) => p.machine); - if (wanted.length === 0) continue; - const onThisWorkstation = (await local( "docker", ["image", "inspect", "--format", "{{.Id}}", requested], 60_000, )).stdout.trim(); @@ -522,46 +518,64 @@ export async function loadHeldImages( ` Build it first (mesh-control's \`make image …\`, or scripts/build-module-runtime.sh).`, ); } + } - // **An image id belongs to the runtime holding it, and does not survive the journey.** It is - // the digest of the image's *configuration*, and a runtime rewrites that configuration as it - // loads: this workstation saves in one format and the machine's older runtime stores it in - // another, so the same bytes arrive under a different name. Measured, not assumed — the same - // image was b86bb81c… here and 2dc21904… on the machine. - // - // So the id is READ BACK from the machine rather than predicted from here. Predicting it is - // what the earlier version did, and it failed at the only useful moment: the manifests would - // have been rewritten to a reference no machine holds, and nothing serves these images, so - // every apply would have stopped at the container with a pull that cannot succeed. - let reference = ""; + // **One archive per machine, not one per image.** + // + // These images are almost entirely the same bytes: the base they share is 227 MB and a module's + // own code is a few. Exported one at a time, that base is written, pushed and loaded once per + // image — for a bed stocking thirty-odd of them, the same 227 MB crosses thirty-odd times, and + // measured on this workstation the whole set came to 9.7 GB. Exported together it is written + // once: two images alone dedupe by 45%, and the saving grows with the count. + // + // This is the slowest thing a raise does, and a four-machine bed has been exceeding its own + // ninety-minute limit while still copying. + const perMachine = new Map(); + for (const requested of scenario.images ?? []) { + for (const plan of plans) { + if (!plan.images.includes(requested)) continue; + const list = perMachine.get(plan.machine) ?? []; + list.push(requested); + perMachine.set(plan.machine, list); + } + } + + // What each machine ended up calling each image, so the disagreement check below still sees + // every machine's answer for every image. + const reported = new Map>(); + + for (const [machine, images] of perMachine) { + const name = machineNames.get(machine); + if (!name || images.length === 0) continue; - // Exported once, handed to each machine that asked for it. The archive is the expensive part - // and it does not depend on the destination. const tar = join(tmpdir(), `mesh-lab-held-${process.pid}-${Date.now()}.tar`); - const saved = await local("docker", ["save", requested, "-o", tar], 900_000); + const saved = await local("docker", ["save", ...images, "-o", tar], 1_800_000); if (!saved.ok) { await unlink(tar).catch(() => {}); - throw new Error(`cannot export ${requested} from this workstation: ${saved.stderr.trim()}`); + throw new Error( + `cannot export ${images.length} image(s) for ${machine} from this workstation: ` + + saved.stderr.trim(), + ); } try { - for (const machine of wanted) { - const name = machineNames.get(machine); - if (!name) continue; - await waitForRuntime(name, machine); - await incus(["file", "push", tar, `${name}/tmp/held.tar`], 900_000); - const loaded = await incusOk( - ["exec", name, "--", "docker", "load", "-i", "/tmp/held.tar"], 900_000, + await waitForRuntime(name, machine); + await incus(["file", "push", tar, `${name}/tmp/held.tar`], 1_800_000); + const loaded = await incusOk( + ["exec", name, "--", "docker", "load", "-i", "/tmp/held.tar"], 1_800_000, + ); + if (!loaded?.includes("Loaded image")) { + throw new PlacementError( + machine, + `${images.length} image(s) were pushed to ${machine} and did not load.\n` + + ` The runtime said: ${loaded?.trim() || "nothing"}`, ); - if (!loaded?.includes("Loaded image")) { - throw new PlacementError( - machine, - `${requested} was pushed to ${machine} and did not load.\n` + - ` The runtime said: ${loaded?.trim() || "nothing"}`, - ); - } - // What this machine calls it, asked of the tag it was just loaded under. This is the - // reference every manifest naming this image will be rewritten to. + } + + for (const requested of images) { + // What this machine calls it, asked of the tag it was just loaded under. An image id is + // the digest of a configuration the receiving runtime rewrites, so it is read back rather + // than predicted from here. const there = (await incusOk( ["exec", name, "--", "docker", "image", "inspect", "--format", "{{.Id}}", requested], 120_000, @@ -575,30 +589,40 @@ export async function loadHeldImages( `itself reports — and there is nothing to fall back to.`, ); } - // **A manifest carries one reference, so the machines must agree on it.** They are built - // from one base image and load one archive, so they do; if that ever stops being true the - // image cannot be named at all from a catalogue, and that is worth stopping for rather - // than rewriting to whichever machine answered last. - if (!reference) { - reference = there; - } else if (reference !== there) { - throw new PlacementError( - machine, - `${requested} is ${there} on ${machine} and ${reference} on a machine already ` + - `loaded.\n` + - ` One manifest cannot name both, and this image exists in no registry to be ` + - `named by instead. The machines' runtimes differ in a way that changes how they ` + - `store what they are given.`, - ); - } - await succeeds(["exec", name, "--", "rm", "-f", "/tmp/held.tar"], 60_000); + const seen = reported.get(requested) ?? new Map(); + seen.set(machine, there); + reported.set(requested, seen); } + await succeeds(["exec", name, "--", "rm", "-f", "/tmp/held.tar"], 60_000); } finally { await unlink(tar).catch(() => {}); } - - held.push({ requested, repository: repositoryOf(requested), reference }); - log(` ${requested} → ${reference.slice(0, 19)}… on ${wanted.join(", ")}`); } + + // **A manifest carries one reference, so the machines must agree on it.** They are built from one + // base and load one archive, so they do; if that ever stops being true the image cannot be named + // from a catalogue at all, and that is worth stopping for rather than rewriting to whichever + // machine answered last. + for (const requested of scenario.images ?? []) { + const seen = reported.get(requested); + if (!seen || seen.size === 0) continue; + const entries = [...seen]; + const first = entries[0]; + if (!first) continue; + const [firstMachine, reference] = first; + for (const [machine, there] of seen) { + if (there === reference) continue; + throw new PlacementError( + machine, + `${requested} is ${there} on ${machine} and ${reference} on ${firstMachine}.\n` + + ` One manifest cannot name both, and this image exists in no registry to be named by ` + + `instead. The machines' runtimes differ in a way that changes how they store what they ` + + `are given.`, + ); + } + held.push({ requested, repository: repositoryOf(requested), reference }); + log(` ${requested} → ${reference.slice(0, 19)}… on ${[...seen.keys()].join(", ")}`); + } + return held; }