novox/hq 04-ISSUES/024. The registry machine had its address set with `ip addr add`; every other machine gets a systemd-networkd unit. That one difference stalled the lab indefinitely. An address set by hand leaves networkd waiting to configure a link it was never told about, so the link sits at `configuring` for ever. `systemd-networkd-wait-online` has TimeoutStartUSec=infinity, so `network-online.target` is never reached — and Docker is ordered after it. `docker load` then blocked on a socket whose daemon was queued behind a target that would never come. Measured before and after on the same scenario: stuck with five pending systemd jobs and `docker` inactive; now `enp5s0 configured`, `docker` active, no jobs, and the whole raise completes in 87.5s. The guess in the issue was wrong, and it was wrong in the usual way — stocking had just been changed, so stocking looked guilty. Stocking takes 34s and always did. Two things that made this cost hours rather than minutes are fixed with it. Placing an image now waits for the container runtime to answer and refuses after 120s naming what systemd is waiting on, so a stall becomes a failure that says why instead of three stacked timeouts totalling 35 minutes. And the end-to-end test passes `onProgress`, so a raise says what step it is on — it printed nothing at all until it finished, which is why 35 minutes of nothing read as a slow test.
205 lines
7.9 KiB
TypeScript
205 lines
7.9 KiB
TypeScript
/**
|
|
* Put the declared addresses on the machines.
|
|
*
|
|
* The lab provides the underlay, and an address is underlay — it is what a hosting
|
|
* provider or a home router would have given the machine before any of our software ran.
|
|
* So the lab assigns it, and the mesh is left to build everything above it.
|
|
*
|
|
* Addresses are applied by matching on MAC rather than interface name. A guest names its
|
|
* interfaces by bus position — `enp5s0`, not `eth0` — so the name the hypervisor uses and
|
|
* the name the guest uses are different, and matching by name silently configures the
|
|
* wrong interface on a multi-homed machine.
|
|
*
|
|
* See novox/hq 02-DECISIONS/0031-the-lab-provides-the-underlay.md
|
|
*/
|
|
|
|
import type { Attachment, Scenario } from "../declaration/types.ts";
|
|
import { incus } from "../incus/client.ts";
|
|
import { macFor } from "./names.ts";
|
|
import { transitAddress } from "./router.ts";
|
|
|
|
export interface Wire {
|
|
device: string;
|
|
mac: string;
|
|
addresses: string[];
|
|
mtu: number | undefined;
|
|
}
|
|
|
|
/**
|
|
* A systemd-networkd unit per interface. Static, because the declaration is the authority:
|
|
* a scenario that let the hypervisor hand out addresses would be the lab supplying facts
|
|
* the declaration is supposed to own.
|
|
*/
|
|
function networkUnit(wire: Wire): string {
|
|
const lines = [
|
|
"[Match]",
|
|
`MACAddress=${wire.mac}`,
|
|
"",
|
|
"[Network]",
|
|
...wire.addresses.map((address) => `Address=${address}`),
|
|
// No gateway and no DNS on purpose. Routing off the segment is a gateway's job, and a
|
|
// scenario that pre-wired it would be arranging what it is supposed to observe.
|
|
"IPv6AcceptRA=no",
|
|
];
|
|
if (wire.mtu !== undefined) {
|
|
lines.push("", "[Link]", `MTUBytes=${wire.mtu}`);
|
|
}
|
|
return lines.join("\n") + "\n";
|
|
}
|
|
|
|
/**
|
|
* Give one link a static address through systemd-networkd, and wait until networkd says it is
|
|
* configured.
|
|
*
|
|
* **Not `ip addr add`**, which is what this replaced and what cost a lab that could not finish.
|
|
* An address set by hand leaves the link `configuring` for ever, because networkd is still
|
|
* waiting to configure something it was never told about. `systemd-networkd-wait-online` then
|
|
* never returns — its timeout is `infinity` — so `network-online.target` is never reached, and
|
|
* **anything ordered after it never starts**. On these machines that is Docker, which meant
|
|
* `docker load` blocked on a socket whose daemon was queued behind a target that would never
|
|
* come. The lab stalled for thirty-five minutes with nothing to say.
|
|
*
|
|
* Every machine already did it this way. The registry did not, and it was the only one that
|
|
* needed Docker before anything else ran.
|
|
*/
|
|
export async function addressLink(
|
|
instanceName: string,
|
|
wire: Wire,
|
|
index = 0,
|
|
): Promise<void> {
|
|
const unit = networkUnit(wire);
|
|
await incus(
|
|
["exec", instanceName, "--", "sh", "-c",
|
|
`mkdir -p /etc/systemd/network && cat > /etc/systemd/network/10-mlab-${index}.network <<'MLAB'\n${unit}MLAB`],
|
|
30_000,
|
|
);
|
|
await incus(["exec", instanceName, "--", "systemctl", "enable", "--now", "systemd-networkd"], 60_000);
|
|
await incus(["exec", instanceName, "--", "systemctl", "restart", "systemd-networkd"], 60_000);
|
|
}
|
|
|
|
export async function applyAddresses(
|
|
scenario: Scenario,
|
|
instanceId: string,
|
|
machineNames: Map<string, string>,
|
|
log: (message: string) => void = () => {},
|
|
): Promise<void> {
|
|
for (const [machine, spec] of Object.entries(scenario.machines)) {
|
|
if (spec.at === "detached") continue;
|
|
const name = machineNames.get(machine);
|
|
if (!name) continue;
|
|
|
|
const wires: Wire[] = [];
|
|
for (const [index, attachment] of spec.at.entries()) {
|
|
wires.push({
|
|
device: `eth${index}`,
|
|
mac: macFor(instanceId, machine, index),
|
|
addresses: attachment.address.map((address) => withPrefix(scenario, attachment.segment, address)),
|
|
mtu: scenario.segments[attachment.segment]?.mtu,
|
|
});
|
|
}
|
|
|
|
for (const [index, wire] of wires.entries()) {
|
|
const unit = networkUnit(wire);
|
|
await incus(
|
|
["exec", name, "--", "sh", "-c",
|
|
`mkdir -p /etc/systemd/network && cat > /etc/systemd/network/10-mlab-${index}.network <<'MLAB'\n${unit}MLAB`],
|
|
30_000,
|
|
);
|
|
}
|
|
|
|
await incus(["exec", name, "--", "systemctl", "enable", "--now", "systemd-networkd"], 60_000);
|
|
await incus(["exec", name, "--", "systemctl", "restart", "systemd-networkd"], 60_000);
|
|
log(` addressed ${machine} (${wires.map((w) => w.addresses.join(",")).join(" | ")})`);
|
|
}
|
|
}
|
|
|
|
/**
|
|
* systemd-networkd wants a prefix length on the address. The declaration gives a bare
|
|
* address and the segment gives the range, so the two are combined here rather than making
|
|
* every scenario repeat the prefix on every machine.
|
|
*/
|
|
function withPrefix(scenario: Scenario, segment: string, address: string): string {
|
|
const ranges = scenario.segments[segment]?.cidr ?? [];
|
|
const wantV6 = address.includes(":");
|
|
for (const range of ranges) {
|
|
const slash = range.lastIndexOf("/");
|
|
if (slash === -1) continue;
|
|
const isV6 = range.slice(0, slash).includes(":");
|
|
if (isV6 === wantV6) return `${address}${range.slice(slash)}`;
|
|
}
|
|
return address;
|
|
}
|
|
|
|
/**
|
|
* Point each machine at the router serving its segment.
|
|
*
|
|
* The route is the machine's, not the mesh's — a default route is what a home network hands
|
|
* out, and a machine that could not reach beyond its own segment would be reproducing the
|
|
* wrong topology. What the lab still does not supply is the overlay: no peers, no hub, no
|
|
* names.
|
|
*
|
|
* The router's inside address is the first host address of the range, chosen rather than
|
|
* declared because a scenario has nothing to say about it.
|
|
*/
|
|
export async function applyDefaultRoutes(
|
|
scenario: Scenario,
|
|
machineNames: Map<string, string>,
|
|
log: (message: string) => void = () => {},
|
|
): Promise<void> {
|
|
for (const [machine, spec] of Object.entries(scenario.machines)) {
|
|
if (spec.at === "detached") continue;
|
|
const name = machineNames.get(machine);
|
|
if (!name) continue;
|
|
|
|
// A machine behind a gateway routes through it. A machine sitting directly on a public
|
|
// segment routes through transit instead — otherwise it can reach its own network and
|
|
// nothing else, which is not what being on the internet means.
|
|
const behind = spec.at.find((a) => scenario.segments[a.segment]?.gateway);
|
|
if (!behind) {
|
|
await routeViaTransit(scenario, spec, name);
|
|
continue;
|
|
}
|
|
|
|
const index = spec.at.indexOf(behind);
|
|
for (const range of scenario.segments[behind.segment]?.cidr ?? []) {
|
|
const slash = range.lastIndexOf("/");
|
|
if (slash === -1) continue;
|
|
const base = range.slice(0, slash);
|
|
const via = base.includes(":")
|
|
? `${base.replace(/::$/, "")}::1`
|
|
: (() => { const o = base.split("."); o[3] = "1"; return o.join("."); })();
|
|
const family = base.includes(":") ? "-6" : "-4";
|
|
await incus(
|
|
["exec", name, "--", "sh", "-c",
|
|
`ip ${family} route replace default via ${via} dev $(ip -o link | awk -F': ' 'NR==${index + 2}{print $2}') 2>/dev/null || true`],
|
|
30_000,
|
|
);
|
|
}
|
|
log(` routed ${machine} via its gateway on ${behind.segment}`);
|
|
}
|
|
}
|
|
|
|
/** A machine on a public segment reaches the other public networks through transit. */
|
|
async function routeViaTransit(
|
|
scenario: Scenario,
|
|
spec: { at: Attachment[] | "detached" },
|
|
name: string,
|
|
): Promise<void> {
|
|
if (spec.at === "detached") return;
|
|
const onPublic = spec.at.find((a) => scenario.segments[a.segment]?.kind === "public");
|
|
if (!onPublic) return;
|
|
|
|
const index = spec.at.indexOf(onPublic);
|
|
for (const cidr of scenario.segments[onPublic.segment]?.cidr ?? []) {
|
|
const via = transitAddress(cidr);
|
|
if (!via) continue;
|
|
const gateway = via.slice(0, via.lastIndexOf("/"));
|
|
const family = gateway.includes(":") ? "-6" : "-4";
|
|
await incus(
|
|
["exec", name, "--", "sh", "-c",
|
|
`ip ${family} route replace default via ${gateway} dev $(ip -o link | awk -F': ' 'NR==${index + 2}{print $2}') 2>/dev/null || true`],
|
|
30_000,
|
|
);
|
|
}
|
|
}
|