diff --git a/scenarios/two-node-db.yml b/scenarios/two-node-db.yml index dc923b6..9295550 100644 --- a/scenarios/two-node-db.yml +++ b/scenarios/two-node-db.yml @@ -48,6 +48,7 @@ images: # its module's provisioner, so no separate mesh-provision-* image is listed — the runtime is the # provisioner (ADR 0048). - mesh-runtime-postgres:development + - mesh-runtime-lavinmq:development - mesh-runtime-redis:development - mesh-runtime-baserow:development - mesh-runtime-letta:development diff --git a/test/integration/assigned-two-node-db.test.ts b/test/integration/assigned-two-node-db.test.ts index d8602f8..e826863 100644 --- a/test/integration/assigned-two-node-db.test.ts +++ b/test/integration/assigned-two-node-db.test.ts @@ -1,19 +1,17 @@ /** - * The DB-consumer chain a single node cannot host, proved across two machines. + * Database consumers on a joined node, provisioned from the ONE foundation store over the overlay. * - * app-postgres and the mesh's own foundation store both want host port 5432, so they cannot share a - * machine. Every earlier catalogue bed put the provider on the same node as the foundation and got - * away with it only because the provider published no 5432 a consumer ever reached, or because the - * foundation's store and the module's postgres were the same container. The moment a real - * postgres PROVIDER must publish 5432 for real consumers to connect, it collides with the store the - * foundation already has there — and the chain is blocked single-node. + * The mesh runs a single postgres — the foundation store, adopted in place as the `postgres` module + * on the control-node (ADR 0078/0079). A module anywhere that requires a database gets one FROM that + * store, not a second postgres of its own; a second would be refused, because the postgres module + * claims the mesh-scoped `mesh-store` seat. * - * This is the split that unblocks it. `anchor` runs the foundation (store, broker, control) and - * NOTHING else. `laptop` runs the whole chain: the postgres and redis PROVIDERS, and the baserow - * and letta CONSUMERS that require them. Provider and consumers are co-located on laptop, so the - * grant never crosses a node boundary and no overlay is needed — only enrolment crosses to anchor, - * over the underlay both machines share. And because the foundation store is on the OTHER node, the - * provider owns laptop's 5432 uncontested. + * This bed proves that across two machines. `anchor` is the control-node: it raises the foundation + * and adopts `postgres` there, so `mesh-store` is the one store and `mesh-postgres` its provisioner. + * `laptop` joins and runs the CONSUMERS — baserow and letta, which require `postgres-database` — plus + * a co-located `redis` (which holds no seat). Each consumer's database is minted on the store on + * anchor and reached over the overlay: their bindings name `anchor.internal`, and their minted logins + * authenticate against the store. redis stays co-located on laptop for baserow's cache. * * The four manifests are the committed catalogue shapes (novox/hq ADR 0039/0047/0048), verbatim * from the catalogue-broad bed — postgres publishes 5432 so its consumers connect, redis runs on @@ -40,6 +38,7 @@ import { test, before, after } from "node:test"; import assert from "node:assert/strict"; import { existsSync, readFileSync } from "node:fs"; +import { dirname, resolve } from "node:path"; import { loadScenario } from "../../src/declaration/parse.ts"; import { raise } from "../../src/lifecycle/raise.ts"; import { destroy, exec } from "../../src/lifecycle/operate.ts"; @@ -63,6 +62,11 @@ const SCENARIO = "two-node-db"; /** The node that carries the whole DB-consumer chain. anchor carries only the foundation. */ const NODE = "laptop"; +const modulesEnv = process.env["MESH_LAB_MODULES"] ?? ""; +const catalogDir = process.env["MESH_LAB_CATALOG"] + ?? (modulesEnv ? resolve(dirname(dirname(dirname(modulesEnv))), "mesh-catalog", "modules") : "") + ?? resolve(process.cwd(), "..", "mesh-catalog", "modules"); + let instanceId = ""; /** The mesh's own images, as the machines hold them. */ let held: HeldImage[] = []; @@ -192,11 +196,12 @@ before(async () => { }, { timeout: 1_800_000 }); after(async () => { + if (process.env["MESH_LAB_KEEP"]) { console.log(`MESH_LAB_KEEP set — leaving ${instanceId} standing`); return; } if (instanceId) await destroy(instanceId); await destroyAll(`${SCENARIO}-`); }, { timeout: 600_000 }); -test("the provider and its consumers ride the second node while the foundation owns 5432 on the first", { +test("consumers on a joined node get their databases from the one foundation store over the overlay", { skip, timeout: 1_500_000, }, async () => { // ================================================================================================ @@ -204,50 +209,6 @@ test("the provider and its consumers ride the second node while the foundation o // they land on changes. // ================================================================================================ - // --- postgres: a database provider whose server publishes 5432 so the real consumers here - // (baserow, letta) reach it at the node address the mesh writes into their grant. ----------------- - const postgresManifest = JSON.stringify({ - module: "postgres", - version: "1", - provides: [{ name: "postgres-database", scope: "mesh" }], - serves: { "postgres-database": { port: 5432 } }, - emits: ["module.postgres.database.provisioned", "module.postgres.database.deprovisioned"], - consumes: ["module.postgres.database.provisioned", "module.postgres.database.deprovisioned"], - receives: { "postgres-database": "/var/lib/postgres/grants/mesh.json" }, - grants: { "postgres-database": "/var/lib/postgres/grants" }, - "own-secrets": { superuser: "/var/lib/postgres/superuser.secret", broker: "/var/lib/mesh/postgres/broker" }, - resources: [ - { id: "mesh-state", type: "directory", path: "/var/lib/mesh/postgres", mode: "0700" }, - { id: "state", type: "directory", path: "/var/lib/postgres", mode: "0700" }, - { id: "grants", type: "directory", path: "/var/lib/postgres/grants", mode: "0700" }, - { id: "superuser-env", type: "file", path: "/var/lib/postgres/superuser.env", mode: "0600", content: "POSTGRES_PASSWORD=${secret:superuser}\n" }, - { id: "data", type: "directory", path: "/services/postgres/db-data", mode: "0700" }, - { id: "net", type: "network", name: "postgres" }, - { - id: "server", type: "container", name: "postgres", image: pinned("postgres"), network: "postgres", - env: { POSTGRES_USER: "postgres", POSTGRES_DB: "postgres" }, - "env-file": ["/var/lib/postgres/superuser.env"], - ports: ["5432"], - volumes: ["/services/postgres/db-data:/var/lib/postgresql/data"], - }, - { - id: "runtime", type: "container", name: "mesh-postgres", image: pinned("mesh-runtime-postgres"), - network: "postgres", - volumes: [ - "/var/lib/mesh/postgres/broker:/run/secrets/broker:ro", - "/var/lib/postgres/grants:/var/lib/postgres/grants:ro", - "/var/lib/postgres/superuser.secret:/run/secrets/superuser:ro", - ], - env: { - MESH_BROKER_FILE: "/run/secrets/broker", - MESH_RECEIVES: "/var/lib/postgres/grants/mesh.json", - MESH_PROVISION_POSTGRES: "postgres://postgres@postgres:5432/postgres?sslmode=disable", - MESH_PROVISION_PASSWORD_FILE: "/run/secrets/superuser", - }, - }, - ], - }); - // --- redis: a cache provider on the host network (127.0.0.1:6379), MESH_SEAL_KEY set lab-locally // because the mesh cannot yet deliver a seal key to a provider's runtime (04-ISSUES). It carries // the committed provides/serves/receives/grants so baserow's redis-cache requirement resolves. ---- @@ -399,44 +360,96 @@ test("the provider and its consumers ride the second node while the foundation o await mesh(`assign ${NODE} ${name}`); } + // Load a committed catalog module.json with its container images rewritten to this scenario's + // pinned digests, so the foundation store can be ADOPTED in place as the one postgres. + function loadManifest(name: string): { manifest: string; broker: boolean } { + const m = JSON.parse(readFileSync(resolve(catalogDir, name, "module.json"), "utf8")) as { + resources?: { type: string; image?: string; artifact?: string }[]; + }; + for (const r of m.resources ?? []) { + if (r.type !== "container") continue; + if (typeof r.image === "string") r.image = pinned(r.image); + else if (typeof r.artifact === "string") { + r.image = pinned(`mesh-runtime-${name}@sha256:${"0".repeat(64)}`); + delete r.artifact; + } + // The provisioner runtime seals a consumer's credential; the mesh cannot yet deliver a seal + // key to a provider's runtime, so set it lab-locally — the same workaround the redis provider + // uses here (hq 04-ISSUES/022, the open provider-seal-key work). + const env = (r as { env?: Record }).env; + if (env && typeof env["MESH_RECEIVES"] === "string") env["MESH_SEAL_KEY"] = "lab-only-seal-key"; + } + const manifest = JSON.stringify(m); + return { manifest, broker: manifest.includes("MESH_BROKER_FILE") }; + } + async function installCatalog(name: string, node: string): Promise { + const { manifest, broker } = loadManifest(name); + await must("anchor", `printf %s ${quote(manifest)} > /tmp/${name}.json && docker cp /tmp/${name}.json mesh-controller:/${name}.json`); + await mesh(`module add /${name}.json`); + if (broker) await mesh(`module issue ${name} --node ${node}`); + await mesh(`assign ${node} ${name}`); + } + // The consumer connects to its provider by the provider's PRIVATE-NETWORK address — the binding's - // `at`, which mesh-controller fills as "where the consuming machine is on the private network, empty - // if it is not on one" (declaration.go). So even though provider and consumer are co-located on - // laptop, the address baserow is handed is the mesh OVERLAY address, and it is empty unless the - // machine is on the overlay. The overlay networking is therefore assigned first, to both nodes. + // `at`, which mesh-controller fills as "where the providing machine is on the private network, + // empty if it is not on one" (declaration.go). The store is on anchor and the consumers on laptop, + // so that address is anchor's OVERLAY name — empty unless both are on the overlay. The overlay + // networking is therefore assigned first, to both nodes. await mesh("overlay place anchor --hub --endpoint 192.0.2.10:51820 --site lab"); await mesh(`overlay place ${NODE} --site lab`); await mesh("assign anchor networking"); await mesh(`assign ${NODE} networking`); - // Providers first, then the consumers that require them. The mesh resolves the whole set at push - // time regardless of order; this order simply reads like the dependency graph. Provider AND - // consumers all go to laptop; the 5432 conflict is gone because the foundation store is on anchor. - await addIssueAssign("postgres", postgresManifest); + // The packet filter on the control-node, so the from:mesh rule admits laptop's consumers to the + // store over the overlay — the firewall half of cross-node provisioning (issue 055). + await installCatalog("nftables", "anchor"); + // Adopt the foundation store AND broker as the ONE postgres and lavinmq, on the control-node: + // each module reconciles the container the foundation raised and brings up its provisioner + // (mesh-postgres mints a database per consumer; mesh-lavinmq a vhost per consumer). There is no + // second store or broker (ADR 0079). Adopting lavinmq also declares the broker's `listens` — so + // its amqps port opens in the firewall's forward chain, which is what lets a consumer on the + // joined node reach the bus cross-node at all. + await installCatalog("postgres", "anchor"); + // The store's superuser is the foundation's, made at genesis (mesh-store's POSTGRES_PASSWORD) — + // carried into the module via `secret accept`, exactly as the genesis bootstrap does. Without it + // the module mints a random superuser that does not match the running store, and the provisioner + // cannot log in to create anyone's database (hq phase3 deliverSuperuser; ADR 0078). + const superPw = (await must("anchor", + `docker inspect mesh-store --format '{{range .Config.Env}}{{println .}}{{end}}' | sed -n 's/^POSTGRES_PASSWORD=//p'`)).trim(); + await must("anchor", `printf %s ${quote(superPw)} > /tmp/superuser && docker cp /tmp/superuser mesh-controller:/superuser`); + await mesh(`secret accept anchor postgres superuser --from /superuser`); + await installCatalog("lavinmq", "anchor"); + await mesh(`push anchor`, 600_000); + await settled("anchor"); + + // The consumers ride laptop; redis is an ordinary co-located provider (it holds no seat). await addIssueAssign("redis", redisManifest); await addIssueAssign("baserow", baserowManifest); await addIssueAssign("letta", lettaManifest); - - // ONE push, ONE convergence — the whole chain resolved and applied together on the second node. await mesh(`push ${NODE}`); await settled(NODE); + // The store's provisioner learns of the cross-node consumers only when anchor is composed again + // — a provision secret is minted composing the CONSUMER, and the provider reads it (issue 057). + await mesh(`push anchor`, 600_000); + await settled("anchor"); + // ================================================================================================ - // THE two-node split — the foundation owns 5432 on anchor, the provider owns it on laptop. + // THE two-node split — the ONE store (adopted) on anchor, its cross-node consumers on laptop. // ================================================================================================ const onAnchor = await must("anchor", `docker ps --format '{{.Names}}'`); const onLaptop = await must(NODE, `docker ps --format '{{.Names}}'`); - assert.match(onAnchor, /(^|\n)mesh-store(\n|$)/, "the foundation store is not on the first node"); - assert.doesNotMatch(onAnchor, /(^|\n)postgres(\n|$)/, - "the postgres provider landed on the foundation node — the 5432 collision this bed exists to avoid"); + assert.match(onAnchor, /(^|\n)mesh-store(\n|$)/, "the foundation store is not on the control-node"); + assert.match(onAnchor, /(^|\n)mesh-postgres(\n|$)/, + "the store's provisioner did not come up on the control-node"); assert.doesNotMatch(onLaptop, /(^|\n)mesh-store(\n|$)/, "the foundation store leaked onto the second node"); - assert.match(onLaptop, /(^|\n)postgres(\n|$)/, "the postgres provider is not on the second node"); + assert.doesNotMatch(onLaptop, /(^|\n)postgres(\n|$)/, + "a second postgres was raised on the second node — the one-store rule (ADR 0079) was violated"); // ================================================================================================ // THE co-residence proof — every module's containers up and stable on the second node. // ================================================================================================ const expected = [ - "postgres", "mesh-postgres", "redis", "mesh-redis", "baserow", "mesh-baserow", "letta", "mesh-letta", @@ -477,22 +490,35 @@ test("the provider and its consumers ride the second node while the foundation o // that letta is a second co-located postgres consumer that gets its OWN credential and reaches its // OWN database (the provider-seal-key gate) — is asserted above (the credential match) and below // (the live provisioning connect); its runtime `mesh-letta` and every other container stay strict. + // + // A provisioner runtime whose provider is cross-node legitimately restarts a few times at + // startup: it exits when the broker is not yet reachable (the overlay tunnel comes up a moment + // after the container does) and docker restarts it until it connects. That is startup churn, not + // a crash-loop — the difference is that churn STOPS. So wait for each container's restart count + // to settle (unchanged over a window) rather than forbid any rise; one that never settles inside + // the deadline is the real crash-loop, and it fails with the trajectory and its logs. const stable = expected.filter((n) => n !== "letta"); - const restarts = new Map(); for (const name of stable) { + const deadline = Date.now() + 300_000; + let prev = -1, stableSince = 0, last = "?"; + while (Date.now() < deadline) { + const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); + last = `running=${running} restarts=${count}`; + const n = Number(count); + if (running === "true" && n === prev) { + if (stableSince === 0) stableSince = Date.now(); + if (Date.now() - stableSince >= 30_000) break; // up and unchanged for 30s — settled + } else { + prev = n; stableSince = 0; + } + await new Promise((r) => setTimeout(r, 5_000)); + } const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); assert.equal(running, "true", - `${name} is not running after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); - restarts.set(name, Number(count)); - } - await new Promise((r) => setTimeout(r, 20000)); - for (const name of stable) { - const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); - assert.equal(running, "true", - `${name} fell over after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); - assert.ok(Number(count) <= (restarts.get(name) ?? 0), - `${name} is crash-looping (restart count rose ${restarts.get(name)} -> ${count}):\n` + - `${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + `${name} is not running after the push (${last}):\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + if (Date.now() >= deadline) + assert.fail(`${name} never stopped restarting within 300s (last ${last}) — a crash-loop, not startup churn:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + void count; } // ================================================================================================ @@ -500,7 +526,7 @@ test("the provider and its consumers ride the second node while the foundation o // reached from laptop over the shared segment) — named for the node that runs it and the module. // ================================================================================================ const users = await must("anchor", `docker exec mesh-broker lavinmqctl list_users 2>&1`); - for (const acct of ["laptop-postgres", "laptop-redis", "laptop-baserow", "laptop-letta"]) { + for (const acct of ["anchor-postgres", "laptop-redis", "laptop-baserow", "laptop-letta"]) { assert.match(users, new RegExp(acct), `the scoped account ${acct} is not on the broker:\n${users}`); } @@ -517,11 +543,13 @@ test("the provider and its consumers ride the second node while the foundation o assert.equal(bound.provision, "postgres-database", `${mod} was bound the wrong provision: ${bound.provision}`); const pw = (await must(NODE, `cat ${secretPath}`)).trim(); assert.ok(bound.as && pw, `${mod}'s login or password was empty (as=${bound.as})`); - const conn = `postgresql://${bound.as}:${encodeURIComponent(pw)}@postgres:5432/${bound.as}?sslmode=disable`; + const conn = `postgresql://${bound.as}:${encodeURIComponent(pw)}@127.0.0.1:5432/${bound.as}?sslmode=disable`; let pg = { out: "", ok: false }; const untilConn = Date.now() + 90_000; while (Date.now() < untilConn) { - pg = await on(NODE, `docker exec mesh-postgres psql ${quote(conn)} -tAc 'select 1' 2>&1`); + // The provisioner (mesh-postgres) is host-networked on anchor beside the store, so it reaches + // it on loopback; the consumer's minted login authenticating there is the cross-node grant working. + pg = await on("anchor", `docker exec mesh-postgres psql ${quote(conn)} -tAc 'select 1' 2>&1`); if (pg.ok && /^1$/m.test(pg.out)) break; if (/authentication failed/i.test(pg.out)) break; await new Promise((r) => setTimeout(r, 3000));