Files
mesh-lab/src/cli.ts
T
jschoubben eba436b6b7 Reach one scenario from the workstation, by name
A scenario is a closed address space: two raised from the same
declaration hold the same addresses and never meet, which is what lets
two run at once and why the lab talks to machines through the
hypervisor rather than over IP. Reaching in from outside breaks that, so
it is opt-in, one scenario at a time, and reversible.

`connect` takes an address on the scenario's public link and writes a
resolver rule answering everything under each machine's name.
`disconnect` gives both back. `connected` says what is true right now,
for somebody who cannot remember.

It refuses rather than guessing when more than one scenario is standing
— the failure being avoided is not an error but one scenario's traffic
arriving in another. It also refuses when a machine's name is already
answered here for something real, because connecting would point that
name at the lab, and the damage would land on the real thing.

Names answer with the segment address rather than the overlay one.
Inside the mesh a name gives a machine's private address; from here that
would need this workstation on the overlay, which is a much larger door.
The segment address reaches the same machine and the same ports, which
is what opening a board in a browser actually needs.

Proven against a live two-node scenario: registry.internal:5000/v2/
answered 200 from this workstation, and so did a wildcard name under the
same machine. Disconnect put the address back, stopped answering, and
left the real mesh's own names alone.

One thing measured rather than assumed: it restarts dnsmasq instead of
reloading it. A reload is SIGHUP, which re-reads the hosts file and
clears the cache but not the configuration — the rule was written, the
reload reported success, and nothing resolved. The daemon's start time
was nine days old afterwards.
2026-09-01 16:42:26 +02:00

328 lines
13 KiB
JavaScript
Executable File

#!/usr/bin/env -S node --experimental-strip-types
/**
* The lab's two callers want different things from the same verbs: a coordinator wants
* structured results and clean teardown, a person wants readable output and the failing
* scenario left standing. Both use these verbs; the difference is what happens after.
*/
import { loadScenario } from "./declaration/parse.ts";
import { DeclarationError } from "./declaration/validate.ts";
import { isReachable, supportedDrivers, pools } from "./incus/client.ts";
import { raise, RaiseError } from "./lifecycle/raise.ts";
import { assertSupported, UnsupportedError } from "./lifecycle/supported.ts";
import { diagramFromDeclaration } from "./diagram/from-declaration.ts";
import { diagramFromLive } from "./diagram/from-live.ts";
import { toDrawio } from "./diagram/drawio.ts";
import { writeFileSync } from "node:fs";
import { destroy, exec, list, restore, snapshot, snapshots } from "./lifecycle/operate.ts";
const USAGE = `mesh-lab — raise a disposable mesh on one machine
check verify this machine can run scenarios
validate <scenario.yml> parse and check a declaration, raising nothing
raise <scenario.yml> materialise it, and wait until the machines are usable
list scenario instances currently standing
exec <instance> <machine> -- <cmd...>
snapshot <instance> <label> capture the whole scenario as one state
restore <instance> <label> return the whole scenario to it
snapshots <instance>
destroy <instance>
diagram <scenario.yml> [out.drawio] draw what a scenario asks for
diagram --live <instance> [out.drawio] draw what is actually raised
warm the scenario kept between runs, and whether it still counts
warm cool destroy it and forget it
suite [paths...] [--no-build] rebuild the artifacts, run the end-to-end tests, leave a receipt
last-run whether the last run still counts; non-zero when it does not
connect [instance] reach the standing scenario from this workstation, by name
disconnect give the address back and stop answering those names
connected what is reachable right now
A scenario is a closed address space, so only one can be reachable at a time: connect refuses
rather than guessing which you meant. It needs root for an address and a resolver rule, and
disconnect puts both back.
Set MESH_LAB_INCUS if the daemon needs a different invocation, e.g. "sudo -n incus".
`;
function fail(message: string): never {
console.error(message);
process.exit(1);
}
/**
* Refuse to run degraded rather than warning. A warning about a slow inner loop is read
* once and ignored forever, and the loop stays slow.
*/
async function check(): Promise<void> {
const problems: string[] = [];
if (!(await isReachable())) {
problems.push(
"the incus daemon is not reachable as this user. If the group was granted recently, " +
"a session that predates it cannot see it — log out and back in, or set MESH_LAB_INCUS.",
);
console.error(problems.map((p) => ` ✗ ${p}`).join("\n"));
process.exit(1);
}
console.log(" ✓ daemon reachable");
const drivers = await supportedDrivers();
const cowDrivers = drivers.filter((d) => d === "btrfs" || d === "zfs");
if (cowDrivers.length === 0) {
problems.push(
"no copy-on-write storage driver is offered. Snapshots would be full copies — " +
"measured at roughly 76x slower, which does not make the lab slow, it makes it unused.",
);
} else {
console.log(` ✓ copy-on-write driver available (${cowDrivers.join(", ")})`);
}
const available = await pools();
const cowPool = available.find((p) => p.driver === "btrfs" || p.driver === "zfs");
if (!cowPool) {
problems.push(
`no pool uses a copy-on-write driver (have: ${available.map((p) => `${p.name}/${p.driver}`).join(", ") || "none"}). ` +
"A pool that exists and is the slow kind is the failure with no symptom.",
);
} else {
console.log(` ✓ copy-on-write pool '${cowPool.name}' (${cowPool.driver})`);
}
if (problems.length > 0) {
console.error(`\n${problems.map((p) => ` ✗ ${p}`).join("\n\n")}`);
process.exit(1);
}
console.log("\nthis machine can run scenarios");
}
async function main(): Promise<void> {
const [verb, ...rest] = process.argv.slice(2);
switch (verb) {
case "check":
return check();
// Rebuilding, running, and recording that it ran are one act. Separate commands would mean a
// run against a stale artifact, or a run nobody recorded — and both are the state
// novox/hq 04-ISSUES/005 is about.
case "suite": {
const { runSuite } = await import("./suite.ts");
process.exitCode = await runSuite(rest);
return;
}
// **What 04-ISSUES/005 says nobody was ever told.** The suite needs a machine with a
// hypervisor, so it cannot run on every push — which means it runs when somebody remembers,
// and remembering is not a mechanism. This asks whether the last run still means anything,
// and exits non-zero when it does not, so a timer or a person can act on it.
case "last-run": {
const { read, judge, whatWasTested } = await import("./lastrun.ts");
const said = judge(read(), new Date(), whatWasTested());
for (const line of said.lines) console.log(line);
if (!said.current) {
console.log("\n `npm run check` runs it.");
process.exitCode = 1;
}
return;
}
// A base state many tests start from, rather than each raising its own mesh.
//
// **The speed is the lesser half.** Tests that share one long-lived mesh accumulate each
// other's state, and a test that reads what the previous one left is a test that passes for
// the wrong reason — which has already happened here once. Returning to a named state between
// tests makes each of them independent.
case "warm": {
const { remembered, ready, cool } = await import("./warm.ts");
const what = rest[0] ?? "status";
if (what === "cool") {
const gone = await cool();
console.log(gone ? `destroyed ${gone}, and forgot it` : "nothing was being kept warm");
return;
}
const held = remembered();
if (!held) {
console.log("nothing is being kept warm.");
console.log(" a scenario is warmed by whatever brought it to a state worth keeping;");
console.log(" the integration suite does it when MESH_LAB_WARM is set.");
return;
}
console.log(`${held.instanceId} — ${held.scenario}, warmed ${held.at}`);
for (const [name, commit] of Object.entries(held.against)) {
console.log(` ${name.padEnd(14)} ${commit}`);
}
const said = await ready(held.scenario);
console.log(said.use === "restore"
? "\n usable: it can be returned to"
: `\n NOT usable: ${said.why}`);
if (said.use !== "restore") process.exitCode = 1;
return;
}
case "base": {
// `base build` exists because a sealed scenario cannot install a container runtime, and
// the runtime has to come from somewhere with a network (novox/hq ADR 0006).
if (rest[0] !== "build") fail("base needs a subcommand: build");
const { buildBaseImage } = await import("./lifecycle/base.ts");
const built = await buildBaseImage((line) => console.log(line));
console.log(`${built.alias}: built, with docker ${built.runtime}`);
return;
}
case "validate": {
const path = rest[0] ?? fail("validate needs a scenario file");
const scenario = loadScenario(path);
console.log(
`${scenario.scenario}: ${Object.keys(scenario.segments).length} segments, ` +
`${Object.keys(scenario.machines).length} machines — valid`,
);
// Valid and raisable are different questions, and a scenario can be the first
// without being the second.
try {
assertSupported(scenario);
} catch (err) {
if (err instanceof UnsupportedError) {
console.log(`\nnot yet raisable:\n - ${err.missing.join("\n - ")}`);
} else throw err;
}
return;
}
case "connect": {
const { connect } = await import("./lifecycle/connect.ts");
const reached = await connect(rest[0]);
console.log(`connected to ${reached.instanceId} as ${reached.address} on ${reached.bridge}\n`);
console.log("these answer here now:");
for (const [machine, address] of Object.entries(reached.machines).sort()) {
console.log(` anything.${machine}.internal → ${address}`);
}
console.log(`\ntry: curl -sI http://${Object.keys(reached.machines)[0]}.internal`);
console.log("run `mesh-lab disconnect` when finished — these names are only true while");
console.log("that scenario is standing.");
return;
}
case "disconnect": {
const { disconnect } = await import("./lifecycle/connect.ts");
for (const line of await disconnect()) console.log(line);
return;
}
case "connected": {
const { connection } = await import("./lifecycle/connect.ts");
const now = await connection();
if (now.length === 0) {
console.log("nothing is connected");
process.exitCode = 1;
return;
}
for (const line of now) console.log(` ${line}`);
return;
}
case "raise": {
const path = rest[0] ?? fail("raise needs a scenario file");
const scenario = loadScenario(path);
const started = Date.now();
const raised = await raise(scenario, {
onProgress: (m) => console.log(m),
...(process.env["MESH_LAB_IMAGE"] ? { image: process.env["MESH_LAB_IMAGE"] } : {}),
});
const seconds = ((Date.now() - started) / 1000).toFixed(1);
console.log(`\nraised ${raised.instanceId} in ${seconds}s — ${raised.machines.length} machines usable`);
if (raised.images.length > 0) {
// Printed because this is what a declaration pins, and it is not knowable until the
// scenario has been raised — the digest belongs to this registry.
console.log(`\nimages served, pinned by digest:`);
for (const image of raised.images) console.log(` ${image}`);
}
if (scenario.snapshot) {
const took = await snapshot(raised.instanceId, scenario.snapshot);
console.log(`snapshot '${scenario.snapshot}' in ${took.toFixed(2)}s`);
}
return;
}
case "list": {
const found = await list();
if (found.length === 0) return console.log("no scenario instances standing");
for (const instance of found) {
console.log(`${instance.instanceId}`);
for (const m of instance.machines) console.log(` ${m.machine.padEnd(16)} ${m.status}`);
}
return;
}
case "exec": {
const [instanceId, machine, ...command] = rest;
if (!instanceId || !machine || command.length === 0) fail("exec <instance> <machine> <cmd...>");
const result = await exec(instanceId, machine, command.filter((c) => c !== "--"));
process.stdout.write(result.stdout);
process.stderr.write(result.stderr);
return;
}
case "snapshot": {
const [instanceId, label] = rest;
if (!instanceId || !label) fail("snapshot <instance> <label>");
const took = await snapshot(instanceId, label);
console.log(`snapshot '${label}' in ${took.toFixed(2)}s`);
return;
}
case "restore": {
const [instanceId, label] = rest;
if (!instanceId || !label) fail("restore <instance> <label>");
const { restoreSeconds, usableSeconds } = await restore(instanceId, label, 180, (m) =>
console.log(m),
);
console.log(
`restored '${label}' in ${restoreSeconds.toFixed(2)}s — usable again after ` +
`${usableSeconds.toFixed(1)}s`,
);
return;
}
case "snapshots": {
const instanceId = rest[0] ?? fail("snapshots <instance>");
const found = await snapshots(instanceId);
console.log(found.length ? found.join("\n") : "none");
return;
}
case "diagram": {
const live = rest[0] === "--live";
const target = (live ? rest[1] : rest[0]) ?? fail("diagram needs a scenario file or --live <instance>");
const diagram = live
? await diagramFromLive(target)
: diagramFromDeclaration(loadScenario(target));
const out = (live ? rest[2] : rest[1]) ?? `${diagram.title}.drawio`;
writeFileSync(out, toDrawio(diagram));
console.log(
`${out} — ${diagram.segments.length} segments, ${diagram.machines.length} machines ` +
`(${diagram.source})`,
);
return;
}
case "destroy": {
const instanceId = rest[0] ?? fail("destroy <instance>");
const { machines, networks } = await destroy(instanceId);
console.log(`destroyed ${instanceId} — ${machines} machines, ${networks} segments`);
return;
}
default:
console.log(USAGE);
process.exit(verb ? 1 : 0);
}
}
main().catch((err: unknown) => {
if (err instanceof DeclarationError || err instanceof RaiseError) fail(err.message);
if (err instanceof UnsupportedError) fail(err.message);
fail(err instanceof Error ? err.message : String(err));
});