The lab raised a `registry` VM, pushed ~73 images into it from the workstation, and rewrote every manifest reference — third-party ones included — to point at it. No production mesh has such a thing. So every bed proved that a machine could fetch an image from a registry that exists nowhere else, and the bootstrap problems that only appear when a machine has to fetch for itself went unfound. What replaces it is the two things that are true in the world: **Public images come from the public internet.** mesh-lab already created a NAT'd uplink for exactly this and attached it to any machine declaring `egress`; no scenario ever declared it. They do now, and third-party references are left exactly as the catalogue writes them. **The mesh's own images have no registry and never will.** mesh-control, mesh-builder, mesh-route-proxy and the per-module runtimes are built from source and exist in no registry. A machine gets them the way an operator's machine does — they are built here and loaded onto it — and is then named by the digest of its own image configuration, which mesh-host now accepts as "an image this machine already holds". `images:` therefore means only *ours*, and a third-party entry is refused rather than quietly loaded: otherwise the fiction returns one convenient line at a time. It is per-machine as well, because "everything, everywhere" was never a description of anything real — handing whole-mesh-full's union to its two 30GiB workstations would fill the disk with runtimes nothing on them will start. **The uplink and the declared gateway would have fought, silently.** A gateway container and the transit router reach the scenario and nothing else; a default route through either is a black hole for anything outside, and it beats the uplink's DHCP route on metric. So a machine with egress states the scenario's ranges explicitly — through the same gateway or transit it would have defaulted to, so the overlay-across-NAT path is unchanged — and leaves the default to the uplink. A range with no path inside the scenario becomes `unreachable` rather than falling through: 192.168.1.0/24 is an ordinary private range in fact, and letting it escape would put scenario traffic on whatever network the workstation is sitting on. `scenarioRoutesFor` is pure and tested, because a decision only a full raise could check is one nobody checks. The registry-reachability check the raise gained earlier is kept, pointed at the real thing: every machine with egress must resolve a name and reach the internet before the raise says it finished. Same failure it was written for — a raise that returns, an apply that dies on its first pull, an instance left a bare shell — now guarding the path that actually carries. The base image's trust of the documentation ranges as plain-HTTP registries STAYS. It was never only for the lab's registry: the mesh has one of its own, the `registry` module, serving artifacts to the whole mesh over plain HTTP from whatever node runs it. Claude-Session: https://claude.ai/code/session_01LrgweAeERJYBg88c5cKDzF
330 lines
13 KiB
JavaScript
Executable File
330 lines
13 KiB
JavaScript
Executable File
#!/usr/bin/env -S node --experimental-strip-types
|
|
/**
|
|
* The lab's two callers want different things from the same verbs: a coordinator wants
|
|
* structured results and clean teardown, a person wants readable output and the failing
|
|
* scenario left standing. Both use these verbs; the difference is what happens after.
|
|
*/
|
|
|
|
import { loadScenario } from "./declaration/parse.ts";
|
|
import { DeclarationError } from "./declaration/validate.ts";
|
|
import { isReachable, supportedDrivers, pools } from "./incus/client.ts";
|
|
import { raise, RaiseError } from "./lifecycle/raise.ts";
|
|
import { assertSupported, UnsupportedError } from "./lifecycle/supported.ts";
|
|
import { diagramFromDeclaration } from "./diagram/from-declaration.ts";
|
|
import { diagramFromLive } from "./diagram/from-live.ts";
|
|
import { toDrawio } from "./diagram/drawio.ts";
|
|
import { writeFileSync } from "node:fs";
|
|
import { destroy, exec, list, restore, snapshot, snapshots } from "./lifecycle/operate.ts";
|
|
|
|
const USAGE = `mesh-lab — raise a disposable mesh on one machine
|
|
|
|
check verify this machine can run scenarios
|
|
validate <scenario.yml> parse and check a declaration, raising nothing
|
|
raise <scenario.yml> materialise it, and wait until the machines are usable
|
|
list scenario instances currently standing
|
|
exec <instance> <machine> -- <cmd...>
|
|
snapshot <instance> <label> capture the whole scenario as one state
|
|
restore <instance> <label> return the whole scenario to it
|
|
snapshots <instance>
|
|
destroy <instance>
|
|
diagram <scenario.yml> [out.drawio] draw what a scenario asks for
|
|
diagram --live <instance> [out.drawio] draw what is actually raised
|
|
|
|
warm the scenario kept between runs, and whether it still counts
|
|
warm cool destroy it and forget it
|
|
suite [paths...] [--no-build] rebuild the artifacts, run the end-to-end tests, leave a receipt
|
|
last-run whether the last run still counts; non-zero when it does not
|
|
|
|
connect [instance] reach the standing scenario from this workstation, by name
|
|
disconnect give the address back and stop answering those names
|
|
connected what is reachable right now
|
|
|
|
A scenario is a closed address space, so only one can be reachable at a time: connect refuses
|
|
rather than guessing which you meant. It needs root for an address and a resolver rule, and
|
|
disconnect puts both back.
|
|
|
|
Set MESH_LAB_INCUS if the daemon needs a different invocation, e.g. "sudo -n incus".
|
|
`;
|
|
|
|
function fail(message: string): never {
|
|
console.error(message);
|
|
process.exit(1);
|
|
}
|
|
|
|
/**
|
|
* Refuse to run degraded rather than warning. A warning about a slow inner loop is read
|
|
* once and ignored forever, and the loop stays slow.
|
|
*/
|
|
async function check(): Promise<void> {
|
|
const problems: string[] = [];
|
|
|
|
if (!(await isReachable())) {
|
|
problems.push(
|
|
"the incus daemon is not reachable as this user. If the group was granted recently, " +
|
|
"a session that predates it cannot see it — log out and back in, or set MESH_LAB_INCUS.",
|
|
);
|
|
console.error(problems.map((p) => ` ✗ ${p}`).join("\n"));
|
|
process.exit(1);
|
|
}
|
|
console.log(" ✓ daemon reachable");
|
|
|
|
const drivers = await supportedDrivers();
|
|
const cowDrivers = drivers.filter((d) => d === "btrfs" || d === "zfs");
|
|
if (cowDrivers.length === 0) {
|
|
problems.push(
|
|
"no copy-on-write storage driver is offered. Snapshots would be full copies — " +
|
|
"measured at roughly 76x slower, which does not make the lab slow, it makes it unused.",
|
|
);
|
|
} else {
|
|
console.log(` ✓ copy-on-write driver available (${cowDrivers.join(", ")})`);
|
|
}
|
|
|
|
const available = await pools();
|
|
const cowPool = available.find((p) => p.driver === "btrfs" || p.driver === "zfs");
|
|
if (!cowPool) {
|
|
problems.push(
|
|
`no pool uses a copy-on-write driver (have: ${available.map((p) => `${p.name}/${p.driver}`).join(", ") || "none"}). ` +
|
|
"A pool that exists and is the slow kind is the failure with no symptom.",
|
|
);
|
|
} else {
|
|
console.log(` ✓ copy-on-write pool '${cowPool.name}' (${cowPool.driver})`);
|
|
}
|
|
|
|
if (problems.length > 0) {
|
|
console.error(`\n${problems.map((p) => ` ✗ ${p}`).join("\n\n")}`);
|
|
process.exit(1);
|
|
}
|
|
console.log("\nthis machine can run scenarios");
|
|
}
|
|
|
|
async function main(): Promise<void> {
|
|
const [verb, ...rest] = process.argv.slice(2);
|
|
|
|
switch (verb) {
|
|
case "check":
|
|
return check();
|
|
|
|
// Rebuilding, running, and recording that it ran are one act. Separate commands would mean a
|
|
// run against a stale artifact, or a run nobody recorded — and both are the state
|
|
// novox/hq 04-ISSUES/005 is about.
|
|
case "suite": {
|
|
const { runSuite } = await import("./suite.ts");
|
|
process.exitCode = await runSuite(rest);
|
|
return;
|
|
}
|
|
|
|
// **What 04-ISSUES/005 says nobody was ever told.** The suite needs a machine with a
|
|
// hypervisor, so it cannot run on every push — which means it runs when somebody remembers,
|
|
// and remembering is not a mechanism. This asks whether the last run still means anything,
|
|
// and exits non-zero when it does not, so a timer or a person can act on it.
|
|
case "last-run": {
|
|
const { read, judge, whatWasTested } = await import("./lastrun.ts");
|
|
const said = judge(read(), new Date(), whatWasTested());
|
|
for (const line of said.lines) console.log(line);
|
|
if (!said.current) {
|
|
console.log("\n `npm run check` runs it.");
|
|
process.exitCode = 1;
|
|
}
|
|
return;
|
|
}
|
|
|
|
// A base state many tests start from, rather than each raising its own mesh.
|
|
//
|
|
// **The speed is the lesser half.** Tests that share one long-lived mesh accumulate each
|
|
// other's state, and a test that reads what the previous one left is a test that passes for
|
|
// the wrong reason — which has already happened here once. Returning to a named state between
|
|
// tests makes each of them independent.
|
|
case "warm": {
|
|
const { remembered, ready, cool } = await import("./warm.ts");
|
|
const what = rest[0] ?? "status";
|
|
if (what === "cool") {
|
|
const gone = await cool();
|
|
console.log(gone ? `destroyed ${gone}, and forgot it` : "nothing was being kept warm");
|
|
return;
|
|
}
|
|
const held = remembered();
|
|
if (!held) {
|
|
console.log("nothing is being kept warm.");
|
|
console.log(" a scenario is warmed by whatever brought it to a state worth keeping;");
|
|
console.log(" the integration suite does it when MESH_LAB_WARM is set.");
|
|
return;
|
|
}
|
|
console.log(`${held.instanceId} — ${held.scenario}, warmed ${held.at}`);
|
|
for (const [name, commit] of Object.entries(held.against)) {
|
|
console.log(` ${name.padEnd(14)} ${commit}`);
|
|
}
|
|
const said = await ready(held.scenario);
|
|
console.log(said.use === "restore"
|
|
? "\n usable: it can be returned to"
|
|
: `\n NOT usable: ${said.why}`);
|
|
if (said.use !== "restore") process.exitCode = 1;
|
|
return;
|
|
}
|
|
|
|
case "base": {
|
|
// `base build` exists because a sealed scenario cannot install a container runtime, and
|
|
// the runtime has to come from somewhere with a network (novox/hq ADR 0006).
|
|
if (rest[0] !== "build") fail("base needs a subcommand: build");
|
|
const { buildBaseImage } = await import("./lifecycle/base.ts");
|
|
const built = await buildBaseImage((line) => console.log(line));
|
|
console.log(`${built.alias}: built, with docker ${built.runtime}`);
|
|
return;
|
|
}
|
|
|
|
case "validate": {
|
|
const path = rest[0] ?? fail("validate needs a scenario file");
|
|
const scenario = loadScenario(path);
|
|
console.log(
|
|
`${scenario.scenario}: ${Object.keys(scenario.segments).length} segments, ` +
|
|
`${Object.keys(scenario.machines).length} machines — valid`,
|
|
);
|
|
// Valid and raisable are different questions, and a scenario can be the first
|
|
// without being the second.
|
|
try {
|
|
assertSupported(scenario);
|
|
} catch (err) {
|
|
if (err instanceof UnsupportedError) {
|
|
console.log(`\nnot yet raisable:\n - ${err.missing.join("\n - ")}`);
|
|
} else throw err;
|
|
}
|
|
return;
|
|
}
|
|
|
|
case "connect": {
|
|
const { connect } = await import("./lifecycle/connect.ts");
|
|
const reached = await connect(rest[0]);
|
|
console.log(`connected to ${reached.instanceId} as ${reached.address} on ${reached.bridge}\n`);
|
|
console.log("these answer here now:");
|
|
for (const [machine, address] of Object.entries(reached.machines).sort()) {
|
|
console.log(` anything.${machine}.internal → ${address}`);
|
|
}
|
|
console.log(`\ntry: curl -sI http://${Object.keys(reached.machines)[0]}.internal`);
|
|
console.log("run `mesh-lab disconnect` when finished — these names are only true while");
|
|
console.log("that scenario is standing.");
|
|
return;
|
|
}
|
|
|
|
case "disconnect": {
|
|
const { disconnect } = await import("./lifecycle/connect.ts");
|
|
for (const line of await disconnect()) console.log(line);
|
|
return;
|
|
}
|
|
|
|
case "connected": {
|
|
const { connection } = await import("./lifecycle/connect.ts");
|
|
const now = await connection();
|
|
if (now.length === 0) {
|
|
console.log("nothing is connected");
|
|
process.exitCode = 1;
|
|
return;
|
|
}
|
|
for (const line of now) console.log(` ${line}`);
|
|
return;
|
|
}
|
|
|
|
case "raise": {
|
|
const path = rest[0] ?? fail("raise needs a scenario file");
|
|
const scenario = loadScenario(path);
|
|
const started = Date.now();
|
|
const raised = await raise(scenario, {
|
|
onProgress: (m) => console.log(m),
|
|
...(process.env["MESH_LAB_IMAGE"] ? { image: process.env["MESH_LAB_IMAGE"] } : {}),
|
|
});
|
|
const seconds = ((Date.now() - started) / 1000).toFixed(1);
|
|
console.log(`\nraised ${raised.instanceId} in ${seconds}s — ${raised.machines.length} machines usable`);
|
|
if (raised.images.length > 0) {
|
|
// Printed because this is what a declaration names them by, and it is not knowable until
|
|
// the image has been built — an image ID is the digest of its own configuration.
|
|
console.log(`\nthe mesh's own images, as the machines now hold them:`);
|
|
for (const image of raised.images) {
|
|
console.log(` ${image.requested} → ${image.reference}`);
|
|
}
|
|
}
|
|
if (scenario.snapshot) {
|
|
const took = await snapshot(raised.instanceId, scenario.snapshot);
|
|
console.log(`snapshot '${scenario.snapshot}' in ${took.toFixed(2)}s`);
|
|
}
|
|
return;
|
|
}
|
|
|
|
case "list": {
|
|
const found = await list();
|
|
if (found.length === 0) return console.log("no scenario instances standing");
|
|
for (const instance of found) {
|
|
console.log(`${instance.instanceId}`);
|
|
for (const m of instance.machines) console.log(` ${m.machine.padEnd(16)} ${m.status}`);
|
|
}
|
|
return;
|
|
}
|
|
|
|
case "exec": {
|
|
const [instanceId, machine, ...command] = rest;
|
|
if (!instanceId || !machine || command.length === 0) fail("exec <instance> <machine> <cmd...>");
|
|
const result = await exec(instanceId, machine, command.filter((c) => c !== "--"));
|
|
process.stdout.write(result.stdout);
|
|
process.stderr.write(result.stderr);
|
|
return;
|
|
}
|
|
|
|
case "snapshot": {
|
|
const [instanceId, label] = rest;
|
|
if (!instanceId || !label) fail("snapshot <instance> <label>");
|
|
const took = await snapshot(instanceId, label);
|
|
console.log(`snapshot '${label}' in ${took.toFixed(2)}s`);
|
|
return;
|
|
}
|
|
|
|
case "restore": {
|
|
const [instanceId, label] = rest;
|
|
if (!instanceId || !label) fail("restore <instance> <label>");
|
|
const { restoreSeconds, usableSeconds } = await restore(instanceId, label, 180, (m) =>
|
|
console.log(m),
|
|
);
|
|
console.log(
|
|
`restored '${label}' in ${restoreSeconds.toFixed(2)}s — usable again after ` +
|
|
`${usableSeconds.toFixed(1)}s`,
|
|
);
|
|
return;
|
|
}
|
|
|
|
case "snapshots": {
|
|
const instanceId = rest[0] ?? fail("snapshots <instance>");
|
|
const found = await snapshots(instanceId);
|
|
console.log(found.length ? found.join("\n") : "none");
|
|
return;
|
|
}
|
|
|
|
case "diagram": {
|
|
const live = rest[0] === "--live";
|
|
const target = (live ? rest[1] : rest[0]) ?? fail("diagram needs a scenario file or --live <instance>");
|
|
const diagram = live
|
|
? await diagramFromLive(target)
|
|
: diagramFromDeclaration(loadScenario(target));
|
|
const out = (live ? rest[2] : rest[1]) ?? `${diagram.title}.drawio`;
|
|
writeFileSync(out, toDrawio(diagram));
|
|
console.log(
|
|
`${out} — ${diagram.segments.length} segments, ${diagram.machines.length} machines ` +
|
|
`(${diagram.source})`,
|
|
);
|
|
return;
|
|
}
|
|
|
|
case "destroy": {
|
|
const instanceId = rest[0] ?? fail("destroy <instance>");
|
|
const { machines, networks } = await destroy(instanceId);
|
|
console.log(`destroyed ${instanceId} — ${machines} machines, ${networks} segments`);
|
|
return;
|
|
}
|
|
|
|
default:
|
|
console.log(USAGE);
|
|
process.exit(verb ? 1 : 0);
|
|
}
|
|
}
|
|
|
|
main().catch((err: unknown) => {
|
|
if (err instanceof DeclarationError || err instanceof RaiseError) fail(err.message);
|
|
if (err instanceof UnsupportedError) fail(err.message);
|
|
fail(err instanceof Error ? err.message : String(err));
|
|
});
|