Reviewed and the criticism was right: 1,072 of 2,128 lines untested, all of it the half that touches the hypervisor, and no gate. The verification I had done was real — pings across NAT, TTL counts, ruleset comparisons — and none of it survived the terminal it ran in, which is 04-ISSUES/005 in miniature. Ten integration tests against a real hypervisor, each named for what it defends. ADR 0031: a raised machine carries no overlay, no wireguard, no mesh config — a scenario that pre-built peering would certify its own work. ADR 0032: exec is the only way in. ADR 0033: routers are containers while machines are virtual machines. And the design's claims: raise waits for usable, snapshots are whole-scenario, NAT hides a private address, published reaches the machine at the gateway's address. Mocking the hypervisor is forbidden, so they skip with a reason on a machine that cannot raise scenarios rather than passing green having checked nothing. The suite earned itself on its first run. It found that a snapshot of a running machine could miss a file written seconds earlier — not stale, absent — because the write was still in the guest's page cache. That is exactly the question the lifecycle design listed as open: does a scenario snapshot need the machines stopped? It does not, but it does need them flushed. snapshot now syncs every machine before capturing, and the design records the answer. The fix buys write-durability, not application-consistency: anything mid-transaction is still captured mid-transaction, and that is now stated rather than assumed. npm run check is the gate — typecheck, 40 unit tests, 10 integration tests.
167 lines
6.5 KiB
TypeScript
167 lines
6.5 KiB
TypeScript
/**
|
|
* What happens to a scenario once it is raised: inspect it, run things in it, capture and
|
|
* return it to a state, and tear it down.
|
|
*
|
|
* Snapshots are WHOLE-SCENARIO. Per-machine would be cheaper and wrong: the mesh keeps
|
|
* state that spans nodes, so restoring one machine to an earlier moment while its peers
|
|
* move on produces a mesh that has never existed and could not. Faults found there would
|
|
* be artefacts of the lab.
|
|
*/
|
|
|
|
import { incus, incusOk, succeeds, taggedInstances, taggedNetworks } from "../incus/client.ts";
|
|
import { machineName } from "./names.ts";
|
|
import { waitUntilAllUsable } from "./ready.ts";
|
|
|
|
export interface Instance {
|
|
instanceId: string;
|
|
machines: { name: string; machine: string; status: string }[];
|
|
}
|
|
|
|
/**
|
|
* Every scenario instance the daemon currently holds, found by the metadata each resource
|
|
* carries rather than by parsing names — a machine called `home-server` would otherwise
|
|
* be split in the wrong place and its instance would appear not to exist.
|
|
*/
|
|
export async function list(): Promise<Instance[]> {
|
|
const byInstance = new Map<string, Instance["machines"]>();
|
|
for (const item of await taggedInstances()) {
|
|
const entry = byInstance.get(item.instanceId) ?? [];
|
|
entry.push({ name: item.name, machine: item.machine, status: item.status });
|
|
byInstance.set(item.instanceId, entry);
|
|
}
|
|
return [...byInstance.entries()]
|
|
.map(([instanceId, machines]) => ({
|
|
instanceId,
|
|
machines: machines.sort((a, b) => a.machine.localeCompare(b.machine)),
|
|
}))
|
|
.sort((a, b) => a.instanceId.localeCompare(b.instanceId));
|
|
}
|
|
|
|
async function machinesOf(instanceId: string): Promise<string[]> {
|
|
const found = (await list()).find((i) => i.instanceId === instanceId);
|
|
if (!found) throw new Error(`no scenario instance '${instanceId}'`);
|
|
return found.machines.map((m) => m.name);
|
|
}
|
|
|
|
/**
|
|
* Run a command on a machine, through incus rather than over IP.
|
|
*
|
|
* A reachability question is therefore asked from INSIDE: *can this machine reach that
|
|
* one* is exec on the first, testing the second. The workstation is not on the scenario's
|
|
* network and its opinion would be a different question with a similar-looking answer.
|
|
*/
|
|
export async function exec(
|
|
instanceId: string,
|
|
machine: string,
|
|
command: string[],
|
|
): Promise<{ stdout: string; stderr: string }> {
|
|
const found = (await taggedInstances()).find(
|
|
(i) => i.instanceId === instanceId && i.machine === machine,
|
|
);
|
|
const name = found?.name ?? machineName(instanceId, machine);
|
|
return incus(["exec", name, "--", ...command], 120_000);
|
|
}
|
|
|
|
/**
|
|
* Capture the whole scenario as one state. Every machine, one name.
|
|
*
|
|
* **Every machine is flushed first, and that is not a precaution.** A snapshot of a running
|
|
* machine captures its disk, not its memory, so a write still sitting in the guest's page
|
|
* cache is simply not in the snapshot. Without the flush a file written seconds earlier can
|
|
* be absent after restore — not stale, absent.
|
|
*
|
|
* Found by the integration test on its first run, which is the question the design listed as
|
|
* open: *does a scenario snapshot need the machines stopped?* It does not, but it does need
|
|
* them flushed.
|
|
*
|
|
* This buys write-durability, not application-consistency. A database mid-transaction is
|
|
* still captured mid-transaction — the snapshot is crash-consistent, and anything needing
|
|
* more has to quiesce itself.
|
|
*/
|
|
export async function snapshot(instanceId: string, label: string): Promise<number> {
|
|
const machines = await machinesOf(instanceId);
|
|
const started = Date.now();
|
|
|
|
for (const name of machines) {
|
|
await incusOk(["exec", name, "--", "sync"], 60_000);
|
|
}
|
|
for (const name of machines) {
|
|
await incus(["snapshot", "create", name, label], 300_000);
|
|
}
|
|
return (Date.now() - started) / 1000;
|
|
}
|
|
|
|
export interface RestoreResult {
|
|
/** How long the restore itself took. */
|
|
restoreSeconds: number;
|
|
/** How long until the scenario was usable again — the number that matters. */
|
|
usableSeconds: number;
|
|
}
|
|
|
|
/**
|
|
* Return the whole scenario to a state. Restoring a subset would produce a mesh that never
|
|
* was, so this is all-or-nothing.
|
|
*
|
|
* Restoring a virtual machine replaces its disk and the machine comes back up, so it
|
|
* reports RUNNING while its agent is still starting — measured, the restore call returns
|
|
* in 0.79s and the very next command fails. Reporting that as "restored" would be
|
|
* transport reported as effect, so this waits for usable and returns both numbers.
|
|
*/
|
|
export async function restore(
|
|
instanceId: string,
|
|
label: string,
|
|
readyTimeoutSeconds = 180,
|
|
log: (message: string) => void = () => {},
|
|
): Promise<RestoreResult> {
|
|
const machines = await machinesOf(instanceId);
|
|
const started = Date.now();
|
|
for (const name of machines) {
|
|
await incus(["snapshot", "restore", name, label], 300_000);
|
|
}
|
|
const restoreSeconds = (Date.now() - started) / 1000;
|
|
|
|
for (const name of machines) {
|
|
await succeeds(["start", name], 60_000);
|
|
}
|
|
await waitUntilAllUsable(machines, readyTimeoutSeconds, log);
|
|
|
|
return { restoreSeconds, usableSeconds: (Date.now() - started) / 1000 };
|
|
}
|
|
|
|
export async function snapshots(instanceId: string): Promise<string[]> {
|
|
const machines = await machinesOf(instanceId);
|
|
const first = machines[0];
|
|
if (!first) return [];
|
|
const csv = (await incusOk(["snapshot", "list", first, "--format", "csv"])) ?? "";
|
|
return csv
|
|
.split("\n")
|
|
.filter(Boolean)
|
|
.map((line) => line.split(",")[0] ?? "")
|
|
.filter(Boolean);
|
|
}
|
|
|
|
/**
|
|
* Tear the instance down: machines first, then the links they were on.
|
|
*
|
|
* Networks are removed last and only if empty — a link still carrying an interface
|
|
* cannot be deleted, and forcing it would leave the daemon with a reference to something
|
|
* gone.
|
|
*/
|
|
export async function destroy(instanceId: string): Promise<{ machines: number; networks: number }> {
|
|
const machines = (await taggedInstances()).filter((i) => i.instanceId === instanceId);
|
|
for (const machine of machines) {
|
|
await succeeds(["delete", "--force", machine.name], 300_000);
|
|
}
|
|
|
|
// Links go last and only once nothing is attached: a network still carrying an
|
|
// interface cannot be deleted, and forcing it would leave a dangling reference.
|
|
let networks = 0;
|
|
for (const network of await taggedNetworks()) {
|
|
if (network.instanceId !== instanceId) continue;
|
|
// `network delete` also succeeds silently — counted with succeeds(), not truthiness.
|
|
if (await succeeds(["network", "delete", network.name], 30_000)) networks++;
|
|
}
|
|
|
|
return { machines: machines.length, networks };
|
|
}
|