/** * Materialise a declaration into a running scenario instance. * * Two rules from the design shape everything here. * * `raise` waits for the machines to be USABLE, not for the calls to return. Measured on a * workstation those are 3.4s and 14.3s apart, and reporting the earlier number would be * the mesh's own recurring failure — transport reported as effect. * * A failed raise LEAVES THE WRECKAGE. Tearing down on failure destroys the only evidence * of what went wrong, and a scenario that failed to raise is more interesting than one * that succeeded. * * See novox/hq 03-DESIGN/01-to-be/03-scenario-lifecycle.md */ import type { Scenario, Segment } from "../declaration/types.ts"; import { incus, incusOk, succeeds, pools, supportedDrivers } from "../incus/client.ts"; import { machineName, macFor, networkName, newInstanceId } from "./names.ts"; import { waitUntilAllUsable } from "./ready.ts"; import { applyAddresses, applyDefaultRoutes } from "./address.ts"; import { assertSupported } from "./supported.ts"; import { planRouters, raiseRouters, raiseTransit } from "./router.ts"; import { applyHostFirewalls } from "./firewall.ts"; import { IMAGE_PREFIX, BASE_IMAGE_ALIAS, BASE_IMAGE_HOWTO, planPlacements, applyPlacements, loadHeldImages } from "./place.ts"; import { baseImageExists, UPSTREAM_IMAGE } from "./base.ts"; import { confirmEgress } from "./egress.ts"; import type { HeldImage } from "../pinning.ts"; import { log as record } from "../log.ts"; /** Drivers whose snapshots are copy-on-write. On `dir` a snapshot is a full copy. */ const COW_DRIVERS = ["btrfs", "zfs"]; export interface RaiseOptions { /** Base image for machines. */ image?: string; /** Reuse an existing instance id rather than minting one — makes raise convergent. */ instanceId?: string; /** Seconds to wait for each machine to become usable. */ readyTimeoutSeconds?: number; onProgress?: (message: string) => void; } export interface RaisedScenario { instanceId: string; scenario: string; machines: string[]; networks: string[]; pool: string; /** * The mesh's own images, as loaded onto the machines, and what a declaration should call them. * * Reported rather than declared: an image built from source has no digest until it has been * built, and what names it here is the digest of its own configuration. * * **Only ours.** Everything third-party is pulled from the internet by the machine that needs * it, so it is not in this list and nothing rewrites it. */ images: HeldImage[]; } export class RaiseError extends Error { readonly instanceId: string; readonly step: string; constructor(instanceId: string, step: string, cause: unknown) { const detail = cause instanceof Error ? cause.message : String(cause); super( `raise failed at '${step}': ${detail}\n` + `The instance '${instanceId}' has been LEFT STANDING for inspection. ` + `Destroy it with: mesh-lab destroy ${instanceId}`, ); this.name = "RaiseError"; this.instanceId = instanceId; this.step = step; } } /** * Pick a pool that can snapshot cheaply, and say so loudly when there is not one. * * Measured: a `dir` snapshot of a 1.5 GB machine takes 9.9s and a full 1.6 GB, with a * second snapshot unfinished after two minutes. On copy-on-write it is 0.13s and costs * the delta. Restoring is the operation the inner loop repeats most, so a `dir` pool does * not make the lab slow — it makes it unused. */ async function choosePool(log: (m: string) => void): Promise { const available = await pools(); const cow = available.find((p) => COW_DRIVERS.includes(p.driver)); if (cow) return cow.name; const drivers = await supportedDrivers(); const possible = drivers.filter((d) => COW_DRIVERS.includes(d)); log( possible.length > 0 ? `WARNING: no copy-on-write pool exists, though the daemon offers ${possible.join("/")}. ` + `Snapshots will be full copies — roughly 76x slower, and the inner loop unusable.` : `WARNING: the daemon offers no copy-on-write driver. Snapshots will be full copies — ` + `roughly 76x slower, and the inner loop unusable. Install btrfs tooling and restart it.`, ); const fallback = available[0]; if (!fallback) throw new Error("no storage pool exists at all"); return fallback.name; } /** * One isolated link per segment. Nothing joins them to anything outside the instance, and * incus is told not to hand out addresses: a scenario declares the underlay, and letting * a hypervisor's DHCP assign addresses would be the lab supplying facts the declaration * is supposed to own. */ async function createNetwork( instanceId: string, segment: string, spec: Segment, ): Promise { const name = networkName(instanceId, segment); if (await succeeds(["network", "show", name], 15_000)) return name; await incus([ "network", "create", name, "ipv4.address=none", "ipv6.address=none", "ipv4.nat=false", "ipv6.nat=false", `user.mesh-lab.instance=${instanceId}`, `user.mesh-lab.segment=${segment}`, // Whether a segment is public and which ranges it carries are facts a link cannot be // asked for afterwards — incus knows only that it is an isolated bridge. Recorded here // so anything reading a raised instance back reads what was applied, rather than // re-opening the declaration and reporting the request as though it were the result. `user.mesh-lab.kind=${spec.kind}`, `user.mesh-lab.cidr=${spec.cidr.join(",")}`, ...(spec.mtu === undefined ? [] : [`user.mesh-lab.mtu=${spec.mtu}`]), ]); return name; } /** * The one network the lab supplies rather than the declaration. * * Every segment a scenario describes is an isolated bridge with no addresses, no DHCP and no NAT, * because the declaration owns addressing. This is the opposite of that on purpose: it is not part * of the scenario, it carries no scenario traffic, and what is routable on it is the host's fact. * * It exists so a machine can fetch what it starts from. A first node pulls three images before * there is any mesh, and the module that gives a mesh its own store pulls one more * (novox/hq 04-ISSUES/029) — none of which anything inside a scenario can serve. * * Tagged like everything else, so tearing the scenario down takes it too. */ async function createUplink(instanceId: string): Promise { const name = networkName(instanceId, "uplink"); if (await succeeds(["network", "show", name], 15_000)) return name; await incus([ "network", "create", name, "ipv4.address=auto", "ipv4.nat=true", "ipv6.address=none", `user.mesh-lab.instance=${instanceId}`, "user.mesh-lab.segment=uplink", ]); return name; } async function createMachine( instanceId: string, machine: string, attachments: { segment: string }[], image: string, pool: string, egress = false, memory = "1GiB", cpus = 2, disk?: string, ): Promise { const name = machineName(instanceId, machine); if (await succeeds(["config", "show", name], 15_000)) return name; const args = [ "init", image, name, "--vm", "-s", pool, // Arch images refuse to boot under secureboot with the shipped keys. Discovered by // the first launch failing with exactly that message. "-c", "security.secureboot=false", "-c", `limits.memory=${memory}`, "-c", `limits.cpu=${cpus}`, "-c", `user.mesh-lab.instance=${instanceId}`, "-c", `user.mesh-lab.machine=${machine}`, ]; // A bigger root disk than the pool default, when the scenario asks — a broad install stocks many // images and exhausts the default, failing the apply with "no space left on device". if (disk) args.push("-d", `root,size=${disk}`); await incus(args, 300_000); // eth0 comes from the profile and points at the wrong network, so every attachment is // explicit. A machine on no segment gets no interface at all — that is what detached is. await succeeds(["config", "device", "remove", name, "eth0"], 15_000); for (const [index, attachment] of attachments.entries()) { await incus([ "config", "device", "add", name, `eth${index}`, "nic", "nictype=bridged", `parent=${networkName(instanceId, attachment.segment)}`, // Explicit, because incus assigns one at runtime without recording it in the device // config — so reading it back returns nothing, and the guest has no stable handle. `hwaddr=${macFor(instanceId, machine, index)}`, ]); } // After every declared attachment, so eth0..ethN keep meaning what the scenario said and the // uplink is whatever comes next. A machine that never asked for one has no such interface at // all, which is the difference between a closed scenario and an open one. if (egress) { await incus([ "config", "device", "add", name, `eth${attachments.length}`, "nic", "nictype=bridged", `parent=${await createUplink(instanceId)}`, `hwaddr=${macFor(instanceId, machine, attachments.length)}`, ]); } return name; } export async function raise( scenario: Scenario, options: RaiseOptions = {}, ): Promise { // **Progress always reaches the file, whether or not anyone asked to see it.** // // This used to be the caller's callback or nothing, and every function below takes its `log` // from here — so a caller that passed none silenced the whole lifecycle. That is exactly what // happened: the end-to-end test called `raise` with no callback, so the one run that mattered // reported not a single step (novox/hq 04-ISSUES/024). // // Teeing rather than replacing: the caller still gets what it asked for, and the record is kept // regardless. A record nobody switched on is the one you want after the thing goes wrong. const log = (message: string): void => { record.info(message); options.onProgress?.(message); }; // A scenario that places a runtime or an image needs machines built from the base image, // because a sealed machine cannot install one (novox/hq ADR 0006). Chosen here rather than // declared, so a scenario says WHAT it needs and not which image provides it. const needsRuntime = (scenario.images ?? []).length > 0 || planPlacements(scenario).some(({ artifacts }) => artifacts.some((a) => a === "runtime" || a.startsWith(IMAGE_PREFIX)) ); if (needsRuntime && !options.image && !(await baseImageExists())) { throw new Error( `this scenario needs a container runtime inside its machines, and '${BASE_IMAGE_ALIAS}' ` + `does not exist.\n${BASE_IMAGE_HOWTO}`, ); } const image = options.image ?? (needsRuntime ? BASE_IMAGE_ALIAS : UPSTREAM_IMAGE); const readyTimeout = options.readyTimeoutSeconds ?? 180; const instanceId = options.instanceId ?? newInstanceId(scenario.scenario, new Date()); // Refuse before spending a minute raising something that would silently lack half of // what it declares. Deliberately outside the try: this is not a raise failure, nothing // has been created, and there is no wreckage to leave standing. assertSupported(scenario); // **The step is the log.** Setting it and recording it are one act, so a step added later // cannot be a step that goes unrecorded — which is the drift that made a thirty-five minute // stall untraceable (novox/hq 04-ISSUES/024). Each entry closes the previous one with its // duration, so the log says where a raise spends its time as well as where it stopped. let step = ""; let stepFrom = Date.now(); const enter = (next: string): string => { if (step) record.info(` ${step} — ${((Date.now() - stepFrom) / 1000).toFixed(1)}s`); record.info(`▶ ${next}`); stepFrom = Date.now(); step = next; return next; }; enter("choosing a storage pool"); try { const pool = await choosePool(log); log(`instance ${instanceId} pool ${pool}`); enter("creating segments"); const networks: string[] = []; for (const [segment, spec] of Object.entries(scenario.segments)) { networks.push(await createNetwork(instanceId, segment, spec)); log(` segment ${segment}`); } enter("creating machines"); const created: string[] = []; const byMachine = new Map(); for (const [machine, spec] of Object.entries(scenario.machines)) { const attachments = spec.at === "detached" ? [] : spec.at; const name = await createMachine( instanceId, machine, attachments, image, pool, spec.at !== "detached" && spec.egress === true, spec.memory, spec.cpus, spec.disk); created.push(name); byMachine.set(machine, name); log(` machine ${machine}${spec.at === "detached" ? " (detached)" : ""}`); } enter("starting machines"); for (const name of created) { await succeeds(["start", name], 60_000); } enter("waiting for machines to become usable"); await waitUntilAllUsable(created, readyTimeout, log); enter("applying declared addresses"); await applyAddresses(scenario, instanceId, byMachine, log); // Transit first: a gateway's default route points at it, so it has to exist. enter("wiring the public networks together"); const transit = await raiseTransit(scenario, instanceId, log); enter("raising routers"); const routers = await raiseRouters(scenario, instanceId, planRouters(scenario, instanceId), log); if (transit) routers.push(transit); enter("routing machines through their gateways"); await applyDefaultRoutes(scenario, byMachine, log); // Last: a machine that refuses inbound must still have been reachable while the lab // was configuring it. enter("applying host firewalls"); await applyHostFirewalls(scenario, byMachine, log); // **Only now is "this machine can reach the outside" a true statement.** The route, the // gateway and the machine's own filtering are all in place, so this is the path a pull takes. // A raise that returned without checking would hand the next step a fact it depends on and // has no way to test — which is how a substrate apply used to die on its first pull. enter("confirming egress reaches the internet"); await confirmEgress(scenario, byMachine, log); // Last, and only once the underlay is real. Placing before the machines can reach each // other would test the host against a network the scenario does not describe. enter("placing"); await applyPlacements(scenario, byMachine, log); // After `placing`, because loading an image needs the container runtime that `placing` // confirmed. The mesh's own images only — everything third-party is pulled by the machine // itself, over its uplink, exactly as it is on a real one. enter("loading the mesh's own images onto the machines"); const images = await loadHeldImages(scenario, byMachine, log); return { instanceId, scenario: scenario.scenario, images, machines: [...created, ...routers], networks, pool, }; } catch (cause) { throw new RaiseError(instanceId, step, cause); } }