/** * What happens to a scenario once it is raised: inspect it, run things in it, capture and * return it to a state, and tear it down. * * Snapshots are WHOLE-SCENARIO. Per-machine would be cheaper and wrong: the mesh keeps * state that spans nodes, so restoring one machine to an earlier moment while its peers * move on produces a mesh that has never existed and could not. Faults found there would * be artefacts of the lab. */ import { incus, incusOk, succeeds, taggedInstances, taggedNetworks } from "../incus/client.ts"; import { machineName } from "./names.ts"; import { waitUntilAllUsable } from "./ready.ts"; export interface Instance { instanceId: string; machines: { name: string; machine: string; status: string }[]; } /** * Every scenario instance the daemon currently holds, found by the metadata each resource * carries rather than by parsing names — a machine called `home-server` would otherwise * be split in the wrong place and its instance would appear not to exist. */ export async function list(): Promise { const byInstance = new Map(); for (const item of await taggedInstances()) { const entry = byInstance.get(item.instanceId) ?? []; entry.push({ name: item.name, machine: item.machine, status: item.status }); byInstance.set(item.instanceId, entry); } return [...byInstance.entries()] .map(([instanceId, machines]) => ({ instanceId, machines: machines.sort((a, b) => a.machine.localeCompare(b.machine)), })) .sort((a, b) => a.instanceId.localeCompare(b.instanceId)); } async function machinesOf(instanceId: string): Promise { const found = (await list()).find((i) => i.instanceId === instanceId); if (!found) throw new Error(`no scenario instance '${instanceId}'`); return found.machines.map((m) => m.name); } /** * Run a command on a machine, through incus rather than over IP. * * A reachability question is therefore asked from INSIDE: *can this machine reach that * one* is exec on the first, testing the second. The workstation is not on the scenario's * network and its opinion would be a different question with a similar-looking answer. */ export async function exec( instanceId: string, machine: string, command: string[], ): Promise<{ stdout: string; stderr: string }> { const found = (await taggedInstances()).find( (i) => i.instanceId === instanceId && i.machine === machine, ); const name = found?.name ?? machineName(instanceId, machine); return incus(["exec", name, "--", ...command], 120_000); } /** * Capture the whole scenario as one state. Every machine, one name. * * **Every machine is flushed first, and that is not a precaution.** A snapshot of a running * machine captures its disk, not its memory, so a write still sitting in the guest's page * cache is simply not in the snapshot. Without the flush a file written seconds earlier can * be absent after restore — not stale, absent. * * Found by the integration test on its first run, which is the question the design listed as * open: *does a scenario snapshot need the machines stopped?* It does not, but it does need * them flushed. * * This buys write-durability, not application-consistency. A database mid-transaction is * still captured mid-transaction — the snapshot is crash-consistent, and anything needing * more has to quiesce itself. */ export async function snapshot(instanceId: string, label: string): Promise { const machines = await machinesOf(instanceId); const started = Date.now(); for (const name of machines) { await incusOk(["exec", name, "--", "sync"], 60_000); } for (const name of machines) { await incus(["snapshot", "create", name, label], 300_000); } return (Date.now() - started) / 1000; } export interface RestoreResult { /** How long the restore itself took. */ restoreSeconds: number; /** How long until the scenario was usable again — the number that matters. */ usableSeconds: number; } /** * Return the whole scenario to a state. Restoring a subset would produce a mesh that never * was, so this is all-or-nothing. * * Restoring a virtual machine replaces its disk and the machine comes back up, so it * reports RUNNING while its agent is still starting — measured, the restore call returns * in 0.79s and the very next command fails. Reporting that as "restored" would be * transport reported as effect, so this waits for usable and returns both numbers. */ export async function restore( instanceId: string, label: string, readyTimeoutSeconds = 180, log: (message: string) => void = () => {}, ): Promise { const machines = await machinesOf(instanceId); const started = Date.now(); for (const name of machines) { await incus(["snapshot", "restore", name, label], 300_000); } const restoreSeconds = (Date.now() - started) / 1000; for (const name of machines) { await succeeds(["start", name], 60_000); } await waitUntilAllUsable(machines, readyTimeoutSeconds, log); return { restoreSeconds, usableSeconds: (Date.now() - started) / 1000 }; } export async function snapshots(instanceId: string): Promise { const machines = await machinesOf(instanceId); const first = machines[0]; if (!first) return []; const csv = (await incusOk(["snapshot", "list", first, "--format", "csv"])) ?? ""; return csv .split("\n") .filter(Boolean) .map((line) => line.split(",")[0] ?? "") .filter(Boolean); } /** * Tear the instance down: machines first, then the links they were on. * * Networks are removed last and only if empty — a link still carrying an interface * cannot be deleted, and forcing it would leave the daemon with a reference to something * gone. */ export async function destroy(instanceId: string): Promise<{ machines: number; networks: number }> { const machines = (await taggedInstances()).filter((i) => i.instanceId === instanceId); for (const machine of machines) { await succeeds(["delete", "--force", machine.name], 300_000); } // Links go last and only once nothing is attached: a network still carrying an // interface cannot be deleted, and forcing it would leave a dangling reference. let networks = 0; for (const network of await taggedNetworks()) { if (network.instanceId !== instanceId) continue; // `network delete` also succeeds silently — counted with succeeds(), not truthiness. if (await succeeds(["network", "delete", network.name], 30_000)) networks++; } return { machines: machines.length, networks }; }