From 033ad7ec6943a7be47b2d7c4309375a94f6f6f14 Mon Sep 17 00:00:00 2001 From: jochen Date: Mon, 31 Aug 2026 15:02:19 +0200 Subject: [PATCH] A run rebuilds what it tests, and leaves a receipt saying what it covered MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The danger is not that the suite breaks. It is that nobody notices it stopped running (novox/hq 04-ISSUES/005). The harness this replaces had not built for two and a half months and nothing said so — and this suite needs a hypervisor, so it inherits exactly that: it runs when somebody remembers, and remembering is not a mechanism. So running, recording, and rebuilding are one act: - the host binary, control-plane image and builder are rebuilt from source first. The last two both parse manifests; building one and not the other left a binary eleven hours old refusing a field the mesh had just renamed, found by a full run. - a receipt lands in XDG state — outside git, because the question is whether *this machine* has run it, and a receipt in git would be a claim about everybody's machine made by whoever committed last. - `last-run` judges it and exits non-zero when it no longer counts. Three faults found by running the thing rather than reading it, each now held by a test confirmed to fail without it: - counted() passed every test while parsing nothing. The runner colours its summary even into a pipe; the fixtures were clean text that had been imagined rather than captured. A fixture that agrees with the mistake proves the mistake. - a receipt for `suite test/lastrun.test.ts` was indistinguishable from one for the real thing — 005's own symptom, rebuilt inside its remedy. The receipt now records what ran. - a tree with uncommitted work reported the bare commit, claiming coverage of code nobody can check out. Nothing else could tell: the hash is identical either way. Proven on real machines: 22/22, against all three repositories. --- README.md | 31 ++++- package.json | 5 +- src/cli.ts | 27 +++++ src/lastrun.ts | 208 ++++++++++++++++++++++++++++++++++ src/rebuild.ts | 85 ++++++++++++++ src/repos.ts | 28 +++++ src/suite.ts | 89 +++++++++++++++ test/integration/mesh.test.ts | 22 ++-- test/lastrun.test.ts | 167 +++++++++++++++++++++++++++ test/rebuild.test.ts | 47 ++++++++ test/suite.test.ts | 68 +++++++++++ 11 files changed, 762 insertions(+), 15 deletions(-) create mode 100644 src/lastrun.ts create mode 100644 src/rebuild.ts create mode 100644 src/repos.ts create mode 100644 src/suite.ts create mode 100644 test/lastrun.test.ts create mode 100644 test/rebuild.test.ts create mode 100644 test/suite.test.ts diff --git a/README.md b/README.md index 9d29145..77fb1eb 100644 --- a/README.md +++ b/README.md @@ -235,13 +235,35 @@ This repository carries implementation. It does not carry decisions. No build step — Node strips the types. ``` -npm test the declaration layer and the diagram, offline, 49 tests -npm run test:integration real scenarios against a real hypervisor, 14 tests +npm test the declaration layer and the diagram, offline +npm run test:integration real scenarios against a real hypervisor npm run typecheck source and tests both — a test that does not compile is a test that silently never ran npm run check typecheck + both suites — this is the gate +npm run last-run when this machine last ran the suite, and whether that still counts ``` +**The integration suite rebuilds what it tests, and leaves a receipt saying it ran.** + +It needs a hypervisor, so it cannot run on every push — which means it runs when somebody +remembers, and *remembering is not a mechanism*. The harness this replaces had not built for two +and a half months and nothing said so (`novox/hq` 04-ISSUES/005). So: + +- **Before the run**, the host binary, the control-plane image and the builder are rebuilt from + source. The last two both parse manifests; building one and not the other is how a rename gets + tested against an eleven-hour-old binary. +- **After the run**, a receipt is written to XDG state — outside the repository, because the + question is *has this machine run it*, and a receipt in git would be a claim about everybody's + machine made by whoever committed last. +- `last-run` judges it and exits non-zero when it no longer counts: old, failed, taken against + commits the repositories have moved past, taken against a tree with uncommitted work, or a run + that never included the end-to-end file. + **A receipt that says nothing about something is not a receipt that clears it.** + +`suite ` runs something narrower, and the receipt records that it did — a green run of the +unit tests must not be readable as coverage of the pipeline. `--no-build` skips the rebuild, for +iterating on a test rather than on the code under it. + **A test names the decision it defends** (`novox/hq` ADR 0017). A decision with no test is one that will quietly stop being true, and nobody learns that from a document: @@ -257,6 +279,11 @@ that will quietly stop being true, and nobody learns that from a document: | the live diagram distinguishes scenery from a node | ADR 0016 | | the live diagram draws what exists, never what was asked for | the diagram design | | a picture nobody can open is not a picture | the diagram design | +| a run that raised no machines is not end-to-end coverage | 04-ISSUES/005 | +| a run whose result could not be read writes nothing | 04-ISSUES/005 | +| the control plane's image and builder are always built together | 04-ISSUES/005 | +| every repository the receipt claims was built by the run | 04-ISSUES/005 | +| a run against uncommitted work does not cover the commit | 04-ISSUES/005 | **Mocking the hypervisor is forbidden.** A fake would assert that the fake behaves as expected, which is the shape of test this project exists to stop shipping. Integration tests skip with a diff --git a/package.json b/package.json index 26b5ed2..553c488 100644 --- a/package.json +++ b/package.json @@ -9,8 +9,9 @@ "scripts": { "typecheck": "tsc --noEmit && tsc --noEmit -p tsconfig.test.json", "test": "node --test --experimental-strip-types 'test/*.test.ts'", - "test:integration": "node --test --test-concurrency=1 --experimental-strip-types 'test/integration/*.test.ts'", - "check": "npm run typecheck && npm test && npm run test:integration" + "test:integration": "node --experimental-strip-types src/cli.ts suite", + "check": "npm run typecheck && npm test && npm run test:integration", + "last-run": "node --experimental-strip-types src/cli.ts last-run" }, "dependencies": { "yaml": "^2.6.0" diff --git a/src/cli.ts b/src/cli.ts index 204e3e6..660fcb0 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -30,6 +30,9 @@ const USAGE = `mesh-lab — raise a disposable mesh on one machine diagram [out.drawio] draw what a scenario asks for diagram --live [out.drawio] draw what is actually raised + suite [paths...] [--no-build] rebuild the artifacts, run the end-to-end tests, leave a receipt + last-run whether the last run still counts; non-zero when it does not + Set MESH_LAB_INCUS if the daemon needs a different invocation, e.g. "sudo -n incus". `; @@ -91,6 +94,30 @@ async function main(): Promise { case "check": return check(); + // Rebuilding, running, and recording that it ran are one act. Separate commands would mean a + // run against a stale artifact, or a run nobody recorded — and both are the state + // novox/hq 04-ISSUES/005 is about. + case "suite": { + const { runSuite } = await import("./suite.ts"); + process.exitCode = await runSuite(rest); + return; + } + + // **What 04-ISSUES/005 says nobody was ever told.** The suite needs a machine with a + // hypervisor, so it cannot run on every push — which means it runs when somebody remembers, + // and remembering is not a mechanism. This asks whether the last run still means anything, + // and exits non-zero when it does not, so a timer or a person can act on it. + case "last-run": { + const { read, judge, whatWasTested } = await import("./lastrun.ts"); + const said = judge(read(), new Date(), whatWasTested()); + for (const line of said.lines) console.log(line); + if (!said.current) { + console.log("\n `npm run check` runs it."); + process.exitCode = 1; + } + return; + } + case "base": { // `base build` exists because a sealed scenario cannot install a container runtime, and // the runtime has to come from somewhere with a network (novox/hq ADR 0006). diff --git a/src/lastrun.ts b/src/lastrun.ts new file mode 100644 index 0000000..9d76fe4 --- /dev/null +++ b/src/lastrun.ts @@ -0,0 +1,208 @@ +/** + * When the end-to-end suite last ran, and against what. + * + * **The fault this exists for is not that the suite breaks — it is that nobody notices it stopped + * running** (novox/hq 04-ISSUES/005). The harness it replaces had not built for two and a half + * months, and nothing said so; the coverage was assumed rather than checked, and several of the + * pipeline's most expensive defects landed inside that window. + * + * This suite is in a better position and the same danger: it needs a machine with a hypervisor, so + * it cannot run on every push, which means it runs when somebody remembers. Remembering is not a + * mechanism. + * + * So a run leaves a receipt, and something can be asked whether the receipt still means anything. + * A receipt that is old, or taken against code the repositories have since moved past, is the + * thing 005 says nobody was ever told. + * + * **Kept outside the repository**, because the question is *has this machine run it* rather than + * *what is committed* — and a receipt in git would be a claim about everyone's machine made by + * whoever committed last. + */ + +import { execFileSync } from "node:child_process"; +import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import { homedir } from "node:os"; +import { dirname, join } from "node:path"; + +import { repositories } from "./repos.ts"; + +/** What a run was taken against, per repository. */ +export type Against = Record; + +export interface Receipt { + /** When it finished, ISO 8601. */ + at: string; + passed: number; + failed: number; + /** The commit each repository was at. Absent for anything that was not a git checkout. */ + against: Against; + /** The test files this run was pointed at. See {@link endToEnd}. */ + ran: string[]; +} + +/** + * endToEnd is the file that raises real machines. A run that did not include it proved nothing + * about the pipeline, however green it was. + * + * The suite takes paths, so it can be pointed at one quick file — and the receipt from that would + * otherwise be indistinguishable from a receipt for the real thing. That is 04-ISSUES/005 again: + * not a suite that fails, a record that says more than the run behind it. + */ +export const endToEnd = "test/integration/mesh.test.ts"; + +/** Where the receipt lives: XDG state, which is for exactly this — data a tool keeps between runs. */ +export function receiptPath(): string { + const state = process.env["XDG_STATE_HOME"] ?? join(homedir(), ".local", "state"); + return join(state, "mesh-lab", "last-run.json"); +} + +/** + * headOf is the commit a directory's repository is at, or "" if it is not one. + * + * **A tree with uncommitted changes is marked, and never equal to the clean commit it sits on.** + * The run tested what was on disk, and that is not what the commit contains — so a receipt naming + * the bare hash would claim coverage of code nobody can check out. Nothing else could tell: the + * hash is identical either way. That is 04-ISSUES/005's overclaim in its quietest form. + */ +export function headOf(directory: string): string { + const git = (args: string[]) => + execFileSync("git", ["-C", directory, ...args], { + encoding: "utf8", + stdio: ["ignore", "pipe", "ignore"], + }); + try { + const head = git(["rev-parse", "--short", "HEAD"]).trim(); + const dirty = git(["status", "--porcelain"]).trim() !== ""; + return dirty ? `${head}+uncommitted` : head; + } catch { + // Not a checkout, or no git. Absent rather than guessed: a receipt claiming a commit it did + // not read is worse than one that says it could not tell. + return ""; + } +} + +/** + * whatWasTested is the repositories this run exercised, by the paths it was given. + * + * From the environment rather than a fixed list, because the paths are how the suite is told what + * to run — so anything it was pointed at is something the receipt should account for, and anything + * it was not pointed at was not tested. + */ +export function whatWasTested(env: NodeJS.ProcessEnv = process.env): Against { + const against: Against = {}; + for (const [name, directory] of Object.entries(repositories(env))) { + const head = headOf(directory); + if (head) against[name] = head; + } + return against; +} + +/** record writes the receipt. Failures are recorded too: a run that failed still ran. */ +export function record( + passed: number, + failed: number, + ran: string[], + env = process.env, +): Receipt { + const receipt: Receipt = { + at: new Date().toISOString(), + passed, + failed, + against: whatWasTested(env), + ran, + }; + const path = receiptPath(); + mkdirSync(dirname(path), { recursive: true }); + writeFileSync(path, JSON.stringify(receipt, null, 2) + "\n"); + return receipt; +} + +/** read returns the receipt, or null when this machine has never run the suite. */ +export function read(): Receipt | null { + try { + return JSON.parse(readFileSync(receiptPath(), "utf8")) as Receipt; + } catch { + return null; + } +} + +export interface Verdict { + /** True when the receipt still says something about the code as it stands. */ + current: boolean; + lines: string[]; +} + +/** + * judge says whether the last run still means anything. + * + * Two ways it can stop meaning something, and they read differently: it was long ago, or the code + * has moved since. The second is the one that matters — a suite that passed against code nobody + * runs any more is coverage in name. + */ +export function judge( + receipt: Receipt | null, + now: Date, + against: Against, + staleAfterDays = 7, +): Verdict { + if (!receipt) { + return { + current: false, + lines: [ + "this machine has never run the end-to-end suite.", + " Nothing here has been checked end to end, which is not the same as nothing being wrong.", + ], + }; + } + + // A receipt written before this field existed says nothing about what it ran, and the honest + // reading of "nothing said" is not "everything". + const endToEndRan = (receipt.ran ?? []).some((path) => path.endsWith(endToEnd)); + + const days = (now.getTime() - Date.parse(receipt.at)) / 86_400_000; + const lines: string[] = []; + const outcome = receipt.failed > 0 + ? `last ran ${ago(days)} and ${receipt.failed} test(s) failed` + : `last passed ${ago(days)}, ${receipt.passed} test(s)`; + lines.push(`the end-to-end suite ${outcome}`); + + let moved = false; + for (const name of Object.keys(against).sort()) { + const then = receipt.against[name]; + const now = against[name]; + if (!then) { + lines.push(` ${name.padEnd(14)} was not accounted for in that run`); + moved = true; + continue; + } + if (then === now) { + lines.push(` ${name.padEnd(14)} at ${then} (unchanged)`); + continue; + } + lines.push(` ${name.padEnd(14)} at ${then}, now at ${now}`); + moved = true; + } + + if (!endToEndRan) { + lines.push(` That run did not include ${endToEnd}, so it raised no machines.`); + } + + if (receipt.failed > 0) { + lines.push(" Nothing has been proven end to end since."); + } else if (moved) { + lines.push(" What it proved was proven about code that has since changed."); + } else if (days > staleAfterDays) { + lines.push(` Nothing has changed since, but that was more than ${staleAfterDays} days ago.`); + } + + return { + current: endToEndRan && receipt.failed === 0 && !moved && days <= staleAfterDays, + lines, + }; +} + +function ago(days: number): string { + if (days < 1 / 24) return "less than an hour ago"; + if (days < 1) return `${Math.round(days * 24)} hour(s) ago`; + return `${Math.round(days)} day(s) ago`; +} diff --git a/src/rebuild.ts b/src/rebuild.ts new file mode 100644 index 0000000..5a73332 --- /dev/null +++ b/src/rebuild.ts @@ -0,0 +1,85 @@ +/** + * Rebuild what the lab runs, from source, before it runs. + * + * **A stale artifact reporting success against old rules is the fault this project keeps writing + * down** (novox/hq 04-ISSUES/005). The lab consumes three artifacts from two repositories, and they + * were rebuilt by hand, one at a time, from memory. A rename in the control plane's catalogue needs + * both the control-plane image *and* the builder binary, because both parse manifests; rebuilding + * one left a binary eleven hours old refusing a field the mesh had just renamed, and cost a full + * run to find out. + * + * In the repository rather than in a shell script beside it, for the reason 005 is about: a step + * that lives in somebody's terminal history runs when they remember, and remembering is not a + * mechanism. + */ + +import { spawnSync } from "node:child_process"; +import { repositories } from "./repos.ts"; + +export interface Build { + /** What it produces, for the log. */ + what: string; + /** The repository root to run in. */ + in: string; + argv: string[]; + env?: NodeJS.ProcessEnv; +} + +/** + * planned is what must be built, given where this run has been pointed. + * + * Derived from the same environment the suite is configured by, so there is one place that says + * where a repository is. A repository this run was not pointed at is not built — and, per + * {@link whatWasTested}, is not claimed in the receipt either. + */ +export function planned(env: NodeJS.ProcessEnv = process.env): Build[] { + const builds: Build[] = []; + const where = repositories(env); + const host = env["MESH_LAB_HOST_BINARY"]; + if (host && where["mesh-host"]) { + builds.push({ + what: "host", + in: where["mesh-host"], + argv: ["go", "build", "-ldflags=-s -w -X main.builtFor=arch", "-o", host, "./cmd/mesh-host"], + env: { CGO_ENABLED: "0" }, + }); + } + const control = where["mesh-control"]; + if (control) { + builds.push({ what: "control plane image", in: control, argv: ["make", "image"] }); + const builder = env["MESH_LAB_BUILDER"]; + if (builder) { + // Both of these parse manifests. Building one and not the other is the eleven-hour-old + // binary above, so they are one step and not two. + builds.push({ + what: "builder", + in: control, + argv: ["go", "build", "-o", builder, "./cmd/mesh-builder"], + }); + } + } + return builds; +} + +/** rebuild runs the plan, and throws on the first failure rather than testing a stale artifact. */ +export function rebuild(env: NodeJS.ProcessEnv = process.env): string[] { + const built: string[] = []; + for (const build of planned(env)) { + const [command, ...args] = build.argv; + const ran = spawnSync(command!, args, { + cwd: build.in, + env: { ...env, ...build.env }, + encoding: "utf8", + }); + if (ran.status !== 0) { + // Loudly, and stopping. A suite that runs anyway is a suite reporting on code that is not + // the code in front of you, which is the whole of 005. + throw new Error( + `could not build the ${build.what}: ${build.argv.join(" ")} in ${build.in}\n\n` + + `${(ran.stderr || ran.stdout || String(ran.error)).trim()}`, + ); + } + built.push(build.what); + } + return built; +} diff --git a/src/repos.ts b/src/repos.ts new file mode 100644 index 0000000..dadabc9 --- /dev/null +++ b/src/repos.ts @@ -0,0 +1,28 @@ +/** + * Where the repositories this run was pointed at are. + * + * **One place**, because two things need it and they must agree: the rebuild builds these, and the + * receipt claims these. A receipt naming a repository the run did not build is exactly the + * false coverage novox/hq 04-ISSUES/005 is about, and it would arrive by nobody's decision — just + * two derivations drifting. + * + * Derived from the environment the suite is configured by, so a repository this run was not + * pointed at is neither built nor claimed. + */ + +import { dirname } from "node:path"; + +export interface Repositories { + /** Absolute path to the repository root, by name. */ + [name: string]: string; +} + +export function repositories(env: NodeJS.ProcessEnv = process.env): Repositories { + const found: Repositories = {}; + found["mesh-lab"] = process.cwd(); + const host = env["MESH_LAB_HOST_BINARY"]; + if (host) found["mesh-host"] = dirname(host); + const modules = env["MESH_LAB_MODULES"]; + if (modules) found["mesh-control"] = dirname(dirname(modules)); + return found; +} diff --git a/src/suite.ts b/src/suite.ts new file mode 100644 index 0000000..8444c8d --- /dev/null +++ b/src/suite.ts @@ -0,0 +1,89 @@ +/** + * Run the end-to-end suite, and leave a receipt saying it ran. + * + * **Here rather than inside the tests, because the tests cannot know their own totals.** Node's + * runner reports them to whatever invoked it, and a test file inventing its own count would be a + * receipt that says whatever the last edit made it say. + * + * Here rather than in a shell script for the same reason the rebuild is: a step that lives in + * somebody's terminal history is a step that runs when they remember (novox/hq 04-ISSUES/005). + */ + +import { spawn } from "node:child_process"; +import { endToEnd, record } from "./lastrun.ts"; +import { rebuild } from "./rebuild.ts"; + +/** counted is what the runner said, or nulls when it said nothing recognisable. */ +export function counted(output: string): { passed: number | null; failed: number | null } { + // The runner's own summary lines, each on a line of its own. Anchored, so a test *named* + // "pass 3" cannot be mistaken for the total — which is not a hypothetical worry in a suite whose + // tests are named in sentences. + // Stripped first: the runner colours its summary even when its stdout is a pipe, so the line is + // "\x1b[34m\u2139 pass 8\x1b[39m" and an anchored pattern never sees the start of it. Found by + // running this against the real runner — the fixture it was first written against was output I + // had imagined, which is a test that agrees with the mistake it was written beside. + const plain = output.replace(/\u001b\[[0-9;]*m/g, ""); + const total = (what: RegExp) => { + const found = plain.match(what); + return found ? Number(found[1]) : null; + }; + return { + passed: total(/^\s*(?:\u2139|#)\s*pass\s+(\d+)\s*$/m), + failed: total(/^\s*(?:\u2139|#)\s*fail\s+(\d+)\s*$/m), + }; +} + +export async function runSuite(args: string[]): Promise { + const ran = args.filter((a) => a !== "--no-build"); + const files = ran.length > 0 ? ran : [endToEnd]; + + if (!args.includes("--no-build")) { + // Before the run, always. The artifacts are built from two other repositories, and a suite + // that tests yesterday's binary reports on code nobody is looking at (novox/hq 04-ISSUES/005). + const built = rebuild(); + if (built.length > 0) console.log(`built: ${built.join(", ")}\n`); + } + + const running = spawn( + process.execPath, + ["--test", "--test-concurrency=1", "--experimental-strip-types", + ...files], + { stdio: ["inherit", "pipe", "inherit"] }, + ); + + let seen = ""; + running.stdout.on("data", (chunk: Buffer) => { + // Passed through as it arrives: a suite that takes a quarter of an hour must not look hung. + process.stdout.write(chunk); + seen += chunk.toString(); + }); + + const code: number = await new Promise((resolve) => { + running.on("close", (c) => resolve(c ?? 1)); + }); + + console.log("\n" + reportOn(counted(seen), (p, f) => record(p, f, files))); + return code; +} + +/** + * reportOn decides whether this run says anything worth recording, and records it if so. + * + * Separated from the spawning so the decision can be tested: **no receipt rather than a guessed + * one** is the rule that keeps the record meaning something, and it was written where nothing + * could check it — which is 04-ISSUES/005 in miniature, inside the fix for it. + */ +export function reportOn( + counts: { passed: number | null; failed: number | null }, + write: (passed: number, failed: number) => { against: Record; ran: string[] }, +): string { + if (counts.passed === null || counts.failed === null) { + // A run whose result could not be read is a run nobody can say anything about. Writing + // "0 failed" because nothing said otherwise is how a green record comes to mean nothing. + return "could not read what the runner reported; no receipt written"; + } + const receipt = write(counts.passed, counts.failed); + const against = Object.entries(receipt.against).map(([n, c]) => `${n} ${c}`).join(", "); + return `recorded: ${receipt.ran.join(", ")} — ${counts.passed} passed, ` + + `${counts.failed} failed, against ${against || "nothing in git"}`; +} diff --git a/test/integration/mesh.test.ts b/test/integration/mesh.test.ts index 5bb3e8f..63f42fc 100644 --- a/test/integration/mesh.test.ts +++ b/test/integration/mesh.test.ts @@ -1396,17 +1396,6 @@ test("a service is reached by a name under the machine it runs on", { await mesh("push"); await new Promise((r) => setTimeout(r, 25_000)); - // `on`, not `must`: `is-active` exits non-zero for a unit that failed, so `must` would throw - // before the assertion below — taking every diagnostic with it. That happened, and the run said - // only "failed". - for (const machine of ["anchor", "laptop"]) { - const state = await on(machine, `systemctl is-active dnsmasq.service`); - if (state.out.trim() === "active") continue; - assert.fail(`the resolver is not running on ${machine} (${state.out.trim()}):\n\n` + - `its config:\n${(await on(machine, `cat /etc/dnsmasq.conf`)).out}\n` + - `${await diagnose(machine)}`); - } - // Everything this test could want to know, gathered in one place. // // Three times now a diagnostic has not run because the thing before it threw: `must` on a @@ -1424,6 +1413,17 @@ test("a service is reached by a name under the machine it runs on", { `asked directly:\n${(await on(machine, `timeout 5 resolvectl query postgres.anchor.internal 2>&1 || echo "no answer"`)).out}`; + // `on`, not `must`: `is-active` exits non-zero for a unit that failed, so `must` would throw + // before the assertion below — taking every diagnostic with it. That happened, and the run said + // only "failed". + for (const machine of ["anchor", "laptop"]) { + const state = await on(machine, `systemctl is-active dnsmasq.service`); + if (state.out.trim() === "active") continue; + assert.fail(`the resolver is not running on ${machine} (${state.out.trim()}):\n\n` + + `its config:\n${(await on(machine, `cat /etc/dnsmasq.conf`)).out}\n` + + `${await diagnose(machine)}`); + } + // Through the machine's own resolver, by the path an application actually takes: nsswitch, then // files, then DNS. `dig` would ask a server directly and prove less — the resolv.conf module is // half of what is being tested, and only this path goes through it. diff --git a/test/lastrun.test.ts b/test/lastrun.test.ts new file mode 100644 index 0000000..3e6d0b9 --- /dev/null +++ b/test/lastrun.test.ts @@ -0,0 +1,167 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { endToEnd, headOf, judge, type Receipt } from "../src/lastrun.ts"; + +const now = new Date("2026-08-31T12:00:00Z"); +const passing = (at: string, against: Record): Receipt => + ({ at, passed: 22, failed: 0, against, ran: [endToEnd] }); + +// A machine that has never run it is told so, rather than told nothing. +// +// novox/hq 04-ISSUES/005: the harness it replaces had not built for two and a half months and +// nothing said so. Silence and success must never look alike. +test("a machine that has never run the suite is told so", () => { + const said = judge(null, now, { "mesh-lab": "aaa" }); + assert.equal(said.current, false); + assert.match(said.lines.join("\n"), /never run/); +}); + +// The one that matters: it passed, and against code nobody runs any more. +test("a run against code that has since changed is not current", () => { + const said = judge( + passing("2026-08-31T11:00:00Z", { "mesh-lab": "aaa", "mesh-control": "bbb" }), + now, + { "mesh-lab": "aaa", "mesh-control": "ccc" }, + ); + assert.equal(said.current, false, "a run against changed code was reported as current"); + const text = said.lines.join("\n"); + assert.match(text, /mesh-control\s+at bbb, now at ccc/, text); + assert.match(text, /code that has since changed/, text); +}); + +// Passing, recent, and against exactly this code is the only thing that counts. +test("a recent run against this code is current", () => { + const said = judge( + passing("2026-08-31T11:00:00Z", { "mesh-lab": "aaa" }), + now, + { "mesh-lab": "aaa" }, + ); + assert.equal(said.current, true, said.lines.join("\n")); + assert.match(said.lines.join("\n"), /unchanged/); +}); + +// Old is a different complaint from moved, and says so — otherwise somebody goes looking for a +// change that did not happen. +test("a run that is merely old says that, not that something changed", () => { + const said = judge( + passing("2026-08-01T11:00:00Z", { "mesh-lab": "aaa" }), + now, + { "mesh-lab": "aaa" }, + ); + assert.equal(said.current, false); + const text = said.lines.join("\n"); + assert.match(text, /Nothing has changed since/, text); + assert.doesNotMatch(text, /has since changed/, text); +}); + +// A failed run is recorded, and does not count as coverage. +test("a run that failed is not coverage", () => { + const said = judge( + { + at: "2026-08-31T11:00:00Z", + passed: 21, + failed: 1, + against: { "mesh-lab": "aaa" }, + ran: [endToEnd], + }, + now, + { "mesh-lab": "aaa" }, + ); + assert.equal(said.current, false); + assert.match(said.lines.join("\n"), /1 test\(s\) failed/); + assert.match(said.lines.join("\n"), /Nothing has been proven end to end since/); +}); + +// A repository the run never accounted for is named, rather than passing silently: a receipt that +// says nothing about something is not a receipt that clears it. +test("a repository the run did not account for is named", () => { + const said = judge( + passing("2026-08-31T11:00:00Z", { "mesh-lab": "aaa" }), + now, + { "mesh-lab": "aaa", "mesh-host": "ddd" }, + ); + assert.equal(said.current, false); + assert.match(said.lines.join("\n"), /mesh-host\s+was not accounted for/); +}); + +// A green run of something else is not a green run of this. +// +// The suite takes paths, so it can be pointed at one quick unit file. Without recording what it +// ran, that receipt and a receipt for the real thing are the same document — which is the whole +// fault of novox/hq 04-ISSUES/005, reintroduced by the fix for it. +test("a run that raised no machines is not end-to-end coverage", () => { + const said = judge( + { + at: "2026-08-31T11:00:00Z", + passed: 6, + failed: 0, + against: { "mesh-lab": "aaa" }, + ran: ["test/lastrun.test.ts"], + }, + now, + { "mesh-lab": "aaa" }, + ); + assert.equal(said.current, false, "a unit run was accepted as end-to-end coverage"); + assert.match(said.lines.join("\n"), /raised no machines/); +}); + +// A receipt written before the mesh recorded what it ran claims nothing, and is read as claiming +// nothing — not as claiming everything. +test("a receipt from before this was recorded is not read as covering everything", () => { + const old = { at: "2026-08-31T11:00:00Z", passed: 22, failed: 0, against: { "mesh-lab": "aaa" } }; + const said = judge(old as unknown as Receipt, now, { "mesh-lab": "aaa" }); + assert.equal(said.current, false); +}); + +// A dirty tree is never equal to the clean commit it sits on. +// +// The run tested what was on disk. Naming the bare hash would claim coverage of code nobody can +// check out — and nothing else could tell, because the hash is identical either way. +test("a run taken against uncommitted work does not count as covering the commit", () => { + const said = judge(passing("2026-08-31T11:00:00Z", { "mesh-lab": "aaa+uncommitted" }), now, { + "mesh-lab": "aaa", + }); + assert.equal(said.current, false, "a run against uncommitted work was read as covering the commit"); + assert.match(said.lines.join("\n"), /aaa\+uncommitted, now at aaa/); +}); + +// headOf against a real repository, because the rule lives in headOf and not in judge. +// +// The first test written for this marked a hand-built receipt and passed with the marking removed +// — it checked how judge reads the value, never that anything produces it. A test that cannot fail +// when the behaviour is deleted is not defending the behaviour. +test("a repository with uncommitted work reports a commit that is marked as such", () => { + const repo = mkdtempSync(join(tmpdir(), "mesh-lab-headof-")); + try { + const git = (...args: string[]) => + execFileSync("git", ["-C", repo, ...args], { stdio: ["ignore", "pipe", "ignore"] }); + git("init", "-q"); + git("config", "user.email", "test@example.invalid"); + git("config", "user.name", "test"); + writeFileSync(join(repo, "a"), "one\n"); + git("add", "a"); + git("commit", "-qm", "first"); + + const clean = headOf(repo); + assert.match(clean, /^[0-9a-f]+$/, `a clean tree was reported as ${clean}`); + + writeFileSync(join(repo, "a"), "two\n"); + assert.equal(headOf(repo), `${clean}+uncommitted`, "an uncommitted change was not marked"); + } finally { + rmSync(repo, { recursive: true, force: true }); + } +}); + +// A directory that is not a checkout is absent from the receipt, not guessed at. +test("a directory that is not a repository reports nothing", () => { + const plain = mkdtempSync(join(tmpdir(), "mesh-lab-plain-")); + try { + assert.equal(headOf(plain), ""); + } finally { + rmSync(plain, { recursive: true, force: true }); + } +}); diff --git a/test/rebuild.test.ts b/test/rebuild.test.ts new file mode 100644 index 0000000..012bb36 --- /dev/null +++ b/test/rebuild.test.ts @@ -0,0 +1,47 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { planned } from "../src/rebuild.ts"; +import { repositories } from "../src/repos.ts"; + +// The control plane's image and the builder are one step, not two. +// +// Both parse manifests. On 2026-08-30 a rename was built into the image and not the binary, and +// the run that found out was a full lab raise. novox/hq 04-ISSUES/005. +test("the control plane's image and builder are always built together", () => { + const builds = planned({ + MESH_LAB_MODULES: "/repo/control/examples/modules", + MESH_LAB_BUILDER: "/repo/control/build/mesh-builder", + }); + const what = builds.map((b) => b.what); + assert.ok(what.includes("control plane image"), "the image was not built"); + assert.ok(what.includes("builder"), "the builder was not built"); + for (const build of builds) assert.equal(build.in, "/repo/control"); +}); + +// A repository this run was not pointed at is not built, and not claimed. +test("only what this run was pointed at is built", () => { + assert.deepEqual(planned({}), []); + const hostOnly = planned({ MESH_LAB_HOST_BINARY: "/repo/host/mesh-host" }); + assert.deepEqual(hostOnly.map((b) => b.what), ["host"]); + assert.equal(hostOnly[0]!.in, "/repo/host"); +}); + +// What the receipt claims and what the run built come from one derivation. +// +// They are separate concerns that must agree: a receipt naming a repository the run did not build +// is false coverage arriving by nobody's decision — just two derivations drifting apart. +// novox/hq 04-ISSUES/005. +test("every repository the receipt claims was built by the run", () => { + const env = { + MESH_LAB_HOST_BINARY: "/repo/host/mesh-host", + MESH_LAB_MODULES: "/repo/control/examples/modules", + MESH_LAB_BUILDER: "/repo/control/build/mesh-builder", + }; + const built = new Set(planned(env).map((b) => b.in)); + for (const [name, directory] of Object.entries(repositories(env))) { + // mesh-lab is the exception, and it is not an omission: it is TypeScript run from source, so + // the code under test *is* the code running. There is nothing to build and nothing to go stale. + if (name === "mesh-lab") continue; + assert.ok(built.has(directory), `${name} (${directory}) is claimed but never built`); + } +}); diff --git a/test/suite.test.ts b/test/suite.test.ts new file mode 100644 index 0000000..a168555 --- /dev/null +++ b/test/suite.test.ts @@ -0,0 +1,68 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { counted, reportOn } from "../src/suite.ts"; + +// The totals come from the runner's own summary, and from nothing else. +test("the runner's totals are read from its summary", () => { + const said = counted("✔ something (1ms)\nℹ tests 22\nℹ pass 22\nℹ fail 0\n"); + assert.deepEqual(said, { passed: 22, failed: 0 }); +}); + +// A test *named* like a total must not be mistaken for one. The summary is a line of its own, and +// the pattern says so — otherwise a test called "pass 3" would rewrite the record. +test("a test named like a total is not a total", () => { + const said = counted("✔ a machine reports pass 3 things (1ms)\nℹ pass 22\nℹ fail 0\n"); + assert.equal(said.passed, 22, "a test name was read as the total"); +}); + +// A failing run is read as a failing run. +test("failures are read", () => { + const said = counted("ℹ pass 21\nℹ fail 1\n"); + assert.deepEqual(said, { passed: 21, failed: 1 }); +}); + +// **No totals is not zero failures.** A run whose result could not be read is a run nobody can say +// anything about, and writing "0 failed" because nothing said otherwise is how a green record +// comes to mean nothing — which is the whole of 04-ISSUES/005. +test("output with no summary yields no totals rather than a clean bill", () => { + const said = counted("the runner crashed before it said anything\n"); + assert.equal(said.passed, null); + assert.equal(said.failed, null); +}); + +// No receipt rather than a guessed one. +// +// The rule that keeps the record meaning something, and it was first written where nothing could +// check it — 04-ISSUES/005 in miniature, inside the fix for it. +test("a run whose result could not be read writes nothing", () => { + let wrote = false; + const said = reportOn({ passed: null, failed: null }, () => { + wrote = true; + return { against: {}, ran: [] }; + }); + assert.equal(wrote, false, "a receipt was written for a run nobody could read"); + assert.match(said, /no receipt written/); +}); + +test("a run that was read is recorded, with what it was read against", () => { + let got: [number, number] | null = null; + const said = reportOn({ passed: 22, failed: 0 }, (p, f) => { + got = [p, f]; + return { against: { "mesh-lab": "abc1234" }, ran: ["test/integration/mesh.test.ts"] }; + }); + assert.deepEqual(got, [22, 0]); + assert.match(said, /mesh\.test\.ts — 22 passed, 0 failed, against mesh-lab abc1234/); +}); + +// The runner's real output, colours and all. +// +// Captured from `node --test` writing into a pipe rather than written by hand: the first version of +// counted() passed every test and read nothing, because the fixtures were clean text and the runner +// emits escape codes. A fixture that agrees with the mistake proves the mistake. +test("the runner's totals are read from output as it actually arrives", () => { + const real = "\u001b[34m\u2139 suites 0\u001b[39m\n" + + "\u001b[34m\u2139 pass 22\u001b[39m\n" + + "\u001b[34m\u2139 fail 0\u001b[39m\n" + + "\u001b[34m\u2139 duration_ms 98.9\u001b[39m\n"; + assert.deepEqual(counted(real), { passed: 22, failed: 0 }); +});