novox/hq 04-ISSUES/024. A run stalled for thirty-five minutes and said nothing. The cause was a link systemd was still configuring, three layers down inside a `docker load` blocked on a socket — and every one of those layers knew what it was waiting for. None of them said so. Three decisions, each doing work. **Every external command is logged, at the three places that run one.** Ninety-seven call sites reach a hypervisor or a container runtime through three wrappers, so instrumenting the wrappers covers all of them and nothing has to remember to log. **A command still running says so while it runs.** A line before and a line after tells you nothing until the after arrives, which is exactly the case that matters. Anything outstanding past fifteen seconds reports itself with how long it has been going. It is reported as still running, not as stuck — which it is is not knowable from there, and a log that calls a slow step a hang teaches people to ignore it. **It goes to a file, written synchronously.** Node block-buffers stdout when redirected and a test runner buffers it again, so a console log can sit minutes behind. `appendFileSync` cannot lag. Two things this found in itself while being written, both the same shape as what it exists to catch: A question that answers no is not a fault. Half the lab's commands are questions — does this network exist, is the agent up yet — and they fail constantly while a scenario comes up. Logging those as faults filled a healthy run with ✗, which is how you end up ignoring ✗ when one is real. They are recorded quietly now, and still recorded. And `around` skipped its own wrapper when a step's level was below the configured one — taking the failure line and the heartbeat with it. The two things worth having at a low level were the two that vanished at exactly the level somebody would use. The gate belongs in `write`. Also unsilences the four call sites that passed a callback throwing everything away, including the one the stall sat in, and tees `raise`'s progress into the file whether or not a caller asked to see it — the end-to-end test passed no callback, so the one run that mattered reported not a single step.
313 lines
13 KiB
TypeScript
313 lines
13 KiB
TypeScript
/**
|
|
* Materialise a declaration into a running scenario instance.
|
|
*
|
|
* Two rules from the design shape everything here.
|
|
*
|
|
* `raise` waits for the machines to be USABLE, not for the calls to return. Measured on a
|
|
* workstation those are 3.4s and 14.3s apart, and reporting the earlier number would be
|
|
* the mesh's own recurring failure — transport reported as effect.
|
|
*
|
|
* A failed raise LEAVES THE WRECKAGE. Tearing down on failure destroys the only evidence
|
|
* of what went wrong, and a scenario that failed to raise is more interesting than one
|
|
* that succeeded.
|
|
*
|
|
* See novox/hq 03-DESIGN/01-to-be/03-scenario-lifecycle.md
|
|
*/
|
|
|
|
import type { Scenario, Segment } from "../declaration/types.ts";
|
|
import { incus, incusOk, succeeds, pools, supportedDrivers } from "../incus/client.ts";
|
|
import { machineName, macFor, networkName, newInstanceId } from "./names.ts";
|
|
import { waitUntilAllUsable } from "./ready.ts";
|
|
import { applyAddresses, applyDefaultRoutes } from "./address.ts";
|
|
import { assertSupported } from "./supported.ts";
|
|
import { planRouters, raiseRouters, raiseTransit } from "./router.ts";
|
|
import { applyHostFirewalls } from "./firewall.ts";
|
|
import { IMAGE_PREFIX, BASE_IMAGE_ALIAS, BASE_IMAGE_HOWTO, planPlacements, applyPlacements } from "./place.ts";
|
|
import { baseImageExists, UPSTREAM_IMAGE } from "./base.ts";
|
|
import { discardStock, raiseRegistry, stockRegistry } from "./registry.ts";
|
|
import { log as record } from "../log.ts";
|
|
|
|
/** Drivers whose snapshots are copy-on-write. On `dir` a snapshot is a full copy. */
|
|
const COW_DRIVERS = ["btrfs", "zfs"];
|
|
|
|
export interface RaiseOptions {
|
|
/** Base image for machines. */
|
|
image?: string;
|
|
/** Reuse an existing instance id rather than minting one — makes raise convergent. */
|
|
instanceId?: string;
|
|
/** Seconds to wait for each machine to become usable. */
|
|
readyTimeoutSeconds?: number;
|
|
onProgress?: (message: string) => void;
|
|
}
|
|
|
|
export interface RaisedScenario {
|
|
instanceId: string;
|
|
scenario: string;
|
|
machines: string[];
|
|
networks: string[];
|
|
pool: string;
|
|
/**
|
|
* Images the scenario's registry serves, as references a declaration can pin.
|
|
*
|
|
* Reported rather than declared, because the digest is the one this registry assigned and
|
|
* is not knowable before it was raised.
|
|
*/
|
|
images: string[];
|
|
}
|
|
|
|
export class RaiseError extends Error {
|
|
readonly instanceId: string;
|
|
readonly step: string;
|
|
|
|
constructor(instanceId: string, step: string, cause: unknown) {
|
|
const detail = cause instanceof Error ? cause.message : String(cause);
|
|
super(
|
|
`raise failed at '${step}': ${detail}\n` +
|
|
`The instance '${instanceId}' has been LEFT STANDING for inspection. ` +
|
|
`Destroy it with: mesh-lab destroy ${instanceId}`,
|
|
);
|
|
this.name = "RaiseError";
|
|
this.instanceId = instanceId;
|
|
this.step = step;
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Pick a pool that can snapshot cheaply, and say so loudly when there is not one.
|
|
*
|
|
* Measured: a `dir` snapshot of a 1.5 GB machine takes 9.9s and a full 1.6 GB, with a
|
|
* second snapshot unfinished after two minutes. On copy-on-write it is 0.13s and costs
|
|
* the delta. Restoring is the operation the inner loop repeats most, so a `dir` pool does
|
|
* not make the lab slow — it makes it unused.
|
|
*/
|
|
async function choosePool(log: (m: string) => void): Promise<string> {
|
|
const available = await pools();
|
|
const cow = available.find((p) => COW_DRIVERS.includes(p.driver));
|
|
if (cow) return cow.name;
|
|
|
|
const drivers = await supportedDrivers();
|
|
const possible = drivers.filter((d) => COW_DRIVERS.includes(d));
|
|
log(
|
|
possible.length > 0
|
|
? `WARNING: no copy-on-write pool exists, though the daemon offers ${possible.join("/")}. ` +
|
|
`Snapshots will be full copies — roughly 76x slower, and the inner loop unusable.`
|
|
: `WARNING: the daemon offers no copy-on-write driver. Snapshots will be full copies — ` +
|
|
`roughly 76x slower, and the inner loop unusable. Install btrfs tooling and restart it.`,
|
|
);
|
|
const fallback = available[0];
|
|
if (!fallback) throw new Error("no storage pool exists at all");
|
|
return fallback.name;
|
|
}
|
|
|
|
/**
|
|
* One isolated link per segment. Nothing joins them to anything outside the instance, and
|
|
* incus is told not to hand out addresses: a scenario declares the underlay, and letting
|
|
* a hypervisor's DHCP assign addresses would be the lab supplying facts the declaration
|
|
* is supposed to own.
|
|
*/
|
|
async function createNetwork(
|
|
instanceId: string,
|
|
segment: string,
|
|
spec: Segment,
|
|
): Promise<string> {
|
|
const name = networkName(instanceId, segment);
|
|
if (await succeeds(["network", "show", name], 15_000)) return name;
|
|
await incus([
|
|
"network", "create", name,
|
|
"ipv4.address=none",
|
|
"ipv6.address=none",
|
|
"ipv4.nat=false",
|
|
"ipv6.nat=false",
|
|
`user.mesh-lab.instance=${instanceId}`,
|
|
`user.mesh-lab.segment=${segment}`,
|
|
// Whether a segment is public and which ranges it carries are facts a link cannot be
|
|
// asked for afterwards — incus knows only that it is an isolated bridge. Recorded here
|
|
// so anything reading a raised instance back reads what was applied, rather than
|
|
// re-opening the declaration and reporting the request as though it were the result.
|
|
`user.mesh-lab.kind=${spec.kind}`,
|
|
`user.mesh-lab.cidr=${spec.cidr.join(",")}`,
|
|
...(spec.mtu === undefined ? [] : [`user.mesh-lab.mtu=${spec.mtu}`]),
|
|
]);
|
|
return name;
|
|
}
|
|
|
|
async function createMachine(
|
|
instanceId: string,
|
|
machine: string,
|
|
attachments: { segment: string }[],
|
|
image: string,
|
|
pool: string,
|
|
): Promise<string> {
|
|
const name = machineName(instanceId, machine);
|
|
if (await succeeds(["config", "show", name], 15_000)) return name;
|
|
|
|
const args = [
|
|
"init", image, name,
|
|
"--vm",
|
|
"-s", pool,
|
|
// Arch images refuse to boot under secureboot with the shipped keys. Discovered by
|
|
// the first launch failing with exactly that message.
|
|
"-c", "security.secureboot=false",
|
|
"-c", "limits.memory=1GiB",
|
|
"-c", "limits.cpu=2",
|
|
"-c", `user.mesh-lab.instance=${instanceId}`,
|
|
"-c", `user.mesh-lab.machine=${machine}`,
|
|
];
|
|
await incus(args, 300_000);
|
|
|
|
// eth0 comes from the profile and points at the wrong network, so every attachment is
|
|
// explicit. A machine on no segment gets no interface at all — that is what detached is.
|
|
await succeeds(["config", "device", "remove", name, "eth0"], 15_000);
|
|
for (const [index, attachment] of attachments.entries()) {
|
|
await incus([
|
|
"config", "device", "add", name, `eth${index}`, "nic",
|
|
"nictype=bridged",
|
|
`parent=${networkName(instanceId, attachment.segment)}`,
|
|
// Explicit, because incus assigns one at runtime without recording it in the device
|
|
// config — so reading it back returns nothing, and the guest has no stable handle.
|
|
`hwaddr=${macFor(instanceId, machine, index)}`,
|
|
]);
|
|
}
|
|
return name;
|
|
}
|
|
|
|
export async function raise(
|
|
scenario: Scenario,
|
|
options: RaiseOptions = {},
|
|
): Promise<RaisedScenario> {
|
|
// **Progress always reaches the file, whether or not anyone asked to see it.**
|
|
//
|
|
// This used to be the caller's callback or nothing, and every function below takes its `log`
|
|
// from here — so a caller that passed none silenced the whole lifecycle. That is exactly what
|
|
// happened: the end-to-end test called `raise` with no callback, so the one run that mattered
|
|
// reported not a single step (novox/hq 04-ISSUES/024).
|
|
//
|
|
// Teeing rather than replacing: the caller still gets what it asked for, and the record is kept
|
|
// regardless. A record nobody switched on is the one you want after the thing goes wrong.
|
|
const log = (message: string): void => {
|
|
record.info(message);
|
|
options.onProgress?.(message);
|
|
};
|
|
|
|
// A scenario that places a runtime or an image needs machines built from the base image,
|
|
// because a sealed machine cannot install one (novox/hq ADR 0006). Chosen here rather than
|
|
// declared, so a scenario says WHAT it needs and not which image provides it.
|
|
const needsRuntime = (scenario.images ?? []).length > 0 ||
|
|
planPlacements(scenario).some(({ artifacts }) =>
|
|
artifacts.some((a) => a === "runtime" || a.startsWith(IMAGE_PREFIX))
|
|
);
|
|
if (needsRuntime && !options.image && !(await baseImageExists())) {
|
|
throw new Error(
|
|
`this scenario needs a container runtime inside its machines, and '${BASE_IMAGE_ALIAS}' ` +
|
|
`does not exist.\n${BASE_IMAGE_HOWTO}`,
|
|
);
|
|
}
|
|
const image = options.image ?? (needsRuntime ? BASE_IMAGE_ALIAS : UPSTREAM_IMAGE);
|
|
const readyTimeout = options.readyTimeoutSeconds ?? 180;
|
|
const instanceId = options.instanceId ?? newInstanceId(scenario.scenario, new Date());
|
|
|
|
// Refuse before spending a minute raising something that would silently lack half of
|
|
// what it declares. Deliberately outside the try: this is not a raise failure, nothing
|
|
// has been created, and there is no wreckage to leave standing.
|
|
assertSupported(scenario);
|
|
|
|
// **The step is the log.** Setting it and recording it are one act, so a step added later
|
|
// cannot be a step that goes unrecorded — which is the drift that made a thirty-five minute
|
|
// stall untraceable (novox/hq 04-ISSUES/024). Each entry closes the previous one with its
|
|
// duration, so the log says where a raise spends its time as well as where it stopped.
|
|
let step = "";
|
|
let stepFrom = Date.now();
|
|
const enter = (next: string): string => {
|
|
if (step) record.info(` ${step} — ${((Date.now() - stepFrom) / 1000).toFixed(1)}s`);
|
|
record.info(`▶ ${next}`);
|
|
stepFrom = Date.now();
|
|
step = next;
|
|
return next;
|
|
};
|
|
|
|
enter("choosing a storage pool");
|
|
try {
|
|
const pool = await choosePool(log);
|
|
log(`instance ${instanceId} pool ${pool}`);
|
|
|
|
enter("creating segments");
|
|
const networks: string[] = [];
|
|
for (const [segment, spec] of Object.entries(scenario.segments)) {
|
|
networks.push(await createNetwork(instanceId, segment, spec));
|
|
log(` segment ${segment}`);
|
|
}
|
|
|
|
enter("creating machines");
|
|
const created: string[] = [];
|
|
const byMachine = new Map<string, string>();
|
|
for (const [machine, spec] of Object.entries(scenario.machines)) {
|
|
const attachments = spec.at === "detached" ? [] : spec.at;
|
|
const name = await createMachine(instanceId, machine, attachments, image, pool);
|
|
created.push(name);
|
|
byMachine.set(machine, name);
|
|
log(` machine ${machine}${spec.at === "detached" ? " (detached)" : ""}`);
|
|
}
|
|
|
|
enter("starting machines");
|
|
for (const name of created) {
|
|
await succeeds(["start", name], 60_000);
|
|
}
|
|
|
|
enter("waiting for machines to become usable");
|
|
await waitUntilAllUsable(created, readyTimeout, log);
|
|
|
|
enter("applying declared addresses");
|
|
await applyAddresses(scenario, instanceId, byMachine, log);
|
|
|
|
// Transit first: a gateway's default route points at it, so it has to exist.
|
|
enter("wiring the public networks together");
|
|
const transit = await raiseTransit(scenario, instanceId, log);
|
|
|
|
enter("raising routers");
|
|
const routers = await raiseRouters(scenario, instanceId, planRouters(scenario, instanceId), log);
|
|
if (transit) routers.push(transit);
|
|
|
|
// Stocked on this workstation, where there is a network, and served from inside the
|
|
// scenario, where there is not (novox/hq 04-ISSUES/009).
|
|
enter("stocking the registry");
|
|
const stock = await stockRegistry(scenario.images ?? [], log);
|
|
let registry: Awaited<ReturnType<typeof raiseRegistry>> = null;
|
|
try {
|
|
enter("raising the registry");
|
|
registry = await raiseRegistry(scenario, instanceId, stock, log);
|
|
} finally {
|
|
// Cleaning up scratch must not fail a raise that succeeded. The scenario is standing
|
|
// and usable; a directory left behind is untidy, and saying so is the honest report.
|
|
try {
|
|
await discardStock(stock);
|
|
} catch (err) {
|
|
log(` (could not remove the registry's scratch directory: ${(err as Error).message})`);
|
|
}
|
|
}
|
|
|
|
enter("routing machines through their gateways");
|
|
await applyDefaultRoutes(scenario, byMachine, log);
|
|
|
|
// Last: a machine that refuses inbound must still have been reachable while the lab
|
|
// was configuring it.
|
|
enter("applying host firewalls");
|
|
await applyHostFirewalls(scenario, byMachine, log);
|
|
|
|
// Last, and only once the underlay is real. Placing before the machines can reach each
|
|
// other would test the host against a network the scenario does not describe.
|
|
enter("placing");
|
|
await applyPlacements(scenario, byMachine, log);
|
|
|
|
return {
|
|
instanceId,
|
|
scenario: scenario.scenario,
|
|
images: registry?.pinned ?? [],
|
|
machines: [...created, ...routers, ...(registry ? [registry.machine] : [])],
|
|
networks,
|
|
pool,
|
|
};
|
|
} catch (cause) {
|
|
throw new RaiseError(instanceId, step, cause);
|
|
}
|
|
}
|