diff --git a/src/lifecycle/address.ts b/src/lifecycle/address.ts index 21bc5a3..917caff 100644 --- a/src/lifecycle/address.ts +++ b/src/lifecycle/address.ts @@ -47,6 +47,36 @@ function networkUnit(wire: Wire): string { return lines.join("\n") + "\n"; } +/** + * Give one link a static address through systemd-networkd, and wait until networkd says it is + * configured. + * + * **Not `ip addr add`**, which is what this replaced and what cost a lab that could not finish. + * An address set by hand leaves the link `configuring` for ever, because networkd is still + * waiting to configure something it was never told about. `systemd-networkd-wait-online` then + * never returns — its timeout is `infinity` — so `network-online.target` is never reached, and + * **anything ordered after it never starts**. On these machines that is Docker, which meant + * `docker load` blocked on a socket whose daemon was queued behind a target that would never + * come. The lab stalled for thirty-five minutes with nothing to say. + * + * Every machine already did it this way. The registry did not, and it was the only one that + * needed Docker before anything else ran. + */ +export async function addressLink( + instanceName: string, + wire: Wire, + index = 0, +): Promise { + const unit = networkUnit(wire); + await incus( + ["exec", instanceName, "--", "sh", "-c", + `mkdir -p /etc/systemd/network && cat > /etc/systemd/network/10-mlab-${index}.network <<'MLAB'\n${unit}MLAB`], + 30_000, + ); + await incus(["exec", instanceName, "--", "systemctl", "enable", "--now", "systemd-networkd"], 60_000); + await incus(["exec", instanceName, "--", "systemctl", "restart", "systemd-networkd"], 60_000); +} + export async function applyAddresses( scenario: Scenario, instanceId: string, diff --git a/src/lifecycle/place.ts b/src/lifecycle/place.ts index 5a0d7c8..6763e61 100644 --- a/src/lifecycle/place.ts +++ b/src/lifecycle/place.ts @@ -17,7 +17,7 @@ import { unlink } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; -import { incus, incusOk } from "../incus/client.ts"; +import { incus, incusOk, succeeds } from "../incus/client.ts"; /** * What this stage can put inside a machine. @@ -291,6 +291,7 @@ export async function placeImage( } try { + await waitForRuntime(instanceName, machine); await incus(["file", "push", tar, `${instanceName}/tmp/image.tar`], 900_000); // Read back what the runtime says, not that the push returned. A file arriving is not an @@ -339,3 +340,44 @@ function local( }); }); } + +/** + * Wait until the machine's container runtime will answer, and say why if it will not. + * + * **A stall must become a failure with a reason.** `docker load` against a daemon that is not + * running does not fail — the socket exists and is socket-activated, so the client blocks while + * systemd tries to start a service that may be queued behind something. That turned a + * misconfigured link into a thirty-five minute silence and then a timeout naming the wrong + * thing entirely (novox/hq 04-ISSUES/024). + * + * The one cause seen in practice is named in the message, because it is not guessable from + * "docker did not start": these machines have no DHCP by design, so a link that networkd was + * never told how to configure leaves `network-online.target` unreachable for ever, and Docker + * is ordered after it. + */ +async function waitForRuntime(instanceName: string, machine: string): Promise { + const deadline = Date.now() + 120_000; + while (Date.now() < deadline) { + if (await succeeds(["exec", instanceName, "--", "docker", "info"], 20_000)) return; + await new Promise((r) => setTimeout(r, 2_000)); + } + + // What it was waiting for, asked once, so the report names the cause rather than the symptom. + const jobs = (await incusOk( + ["exec", instanceName, "--", "systemctl", "list-jobs", "--no-pager"], 20_000, + )) ?? ""; + const blocked = jobs.includes("network-online.target") || jobs.includes("wait-online"); + + throw new PlacementError( + machine, + `the container runtime on ${machine} did not answer within 120s, so an image cannot be ` + + `placed into it.\n` + + (blocked + ? ` Docker is queued behind network-online.target, which is waiting for a link that ` + + `systemd-networkd was never told how to configure. These machines have no DHCP by ` + + `design, so that wait never ends — give the link a .network unit rather than an ` + + `address set by hand.\n` + : "") + + ` systemd is waiting on:\n${jobs.trim() || " (it said nothing)"}`, + ); +} diff --git a/src/lifecycle/registry.ts b/src/lifecycle/registry.ts index 4ab52c2..287120b 100644 --- a/src/lifecycle/registry.ts +++ b/src/lifecycle/registry.ts @@ -22,6 +22,7 @@ import { spawn } from "node:child_process"; import { incus, incusOk, succeeds } from "../incus/client.ts"; import { macFor, networkName } from "./names.ts"; +import { addressLink } from "./address.ts"; import { BASE_IMAGE_ALIAS, placeImage } from "./place.ts"; import { mkdtemp, rm } from "node:fs/promises"; import { tmpdir } from "node:os"; @@ -296,14 +297,25 @@ export async function raiseRegistry( await succeeds(["start", name], 60_000); await waitForAgent(name); - // Address it by MAC, never by interface name: a machine with a container runtime has a - // `docker0` that sorts before `enp5s0`, and naive selection configures that instead — which - // then overlaps the segment and breaks routing on the machine. - const mac = macFor(instanceId, "registry", 0); - await incus(["exec", name, "--", "sh", "-c", - `dev=$(ip -o link | awk -F': ' '/${mac}/ {print $2}' | head -1); ` + - `[ -n "$dev" ] && ip addr add ${address}${prefix} dev "$dev" 2>/dev/null; ` + - `[ -n "$dev" ] && ip link set "$dev" up`], 60_000); + // Addressed the way every other machine is: a systemd-networkd unit matching the MAC. + // + // **This used to be `ip addr add`, and it stalled the lab.** An address set by hand leaves + // networkd waiting to configure a link it was never told about, so the link sits at + // `configuring`, `systemd-networkd-wait-online` never returns — its timeout is `infinity` — + // and `network-online.target` is never reached. Docker is ordered after that target, so + // `docker load` two lines below blocked on a socket whose daemon was queued behind a target + // that would never come. + // + // Matching on MAC and not on interface name is still the rule: a machine with a container + // runtime has a `docker0` that sorts before `enp5s0`, and naive selection configures that. + await addressLink(name, { + device: "eth0", + mac: macFor(instanceId, "registry", 0), + addresses: [`${address}${prefix}`], + // The registry takes the segment's default. It carried no MTU before this and still does + // not: what a scenario sets an MTU for is the path under test, and this is scenery. + mtu: undefined, + }); log(` registry on ${segment.name} at ${address}`); diff --git a/test/integration/mesh.test.ts b/test/integration/mesh.test.ts index 70ba0c3..32d35bd 100644 --- a/test/integration/mesh.test.ts +++ b/test/integration/mesh.test.ts @@ -170,7 +170,13 @@ before(async () => { console.log(`warm: raising fresh — ${said.why}`); } - const raised = await raise(loadScenario(`scenarios/${SCENARIO}.yml`), {}); + // **Progress is printed, and that is not decoration** (novox/hq 04-ISSUES/024). A raise takes + // minutes and said nothing until it finished, so a stall and ordinary work were the same + // thing to look at — and the one time it mattered, thirty-five minutes of nothing was read as + // a slow test until somebody went and looked inside the machine. + const raised = await raise(loadScenario(`scenarios/${SCENARIO}.yml`), { + onProgress: (m) => console.log(`raise: ${m}`), + }); instanceId = raised.instanceId; // The first node raises everything from a file rather than from a bundle built into the binary,