From 0e382f1db7753ca46c6edaae3dab1a3410ccb243 Mon Sep 17 00:00:00 2001 From: jochen Date: Thu, 17 Sep 2026 09:53:35 +0200 Subject: [PATCH] =?UTF-8?q?two-node-db=20is=20GREEN=20on=20the=20one-store?= =?UTF-8?q?=20model=20=E2=80=94=20store=20cross-node=20proven=20end-to-end?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The complete recipe, found across five runs: adopt BOTH the store (postgres) and broker (lavinmq) on the control-node — the broker's `listens` is what opens 5671 in the firewall for cross-node bus access; deliver the store's genesis superuser via `secret accept` (else the module mints a random one that cannot log in to the running store); inject the provider seal key (open hq issue 022 workaround); push the provider node again after the remote consumers (issue 057); and tolerate provisioner-runtime startup churn — a cross-node provisioner exits until the overlay tunnel is up, then settles. baserow and letta on the joined node get their databases from the one foundation store over the overlay. https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx --- scenarios/two-node-db.yml | 1 + test/integration/assigned-two-node-db.test.ts | 61 ++++++++++++++----- 2 files changed, 47 insertions(+), 15 deletions(-) diff --git a/scenarios/two-node-db.yml b/scenarios/two-node-db.yml index dc923b6..9295550 100644 --- a/scenarios/two-node-db.yml +++ b/scenarios/two-node-db.yml @@ -48,6 +48,7 @@ images: # its module's provisioner, so no separate mesh-provision-* image is listed — the runtime is the # provisioner (ADR 0048). - mesh-runtime-postgres:development + - mesh-runtime-lavinmq:development - mesh-runtime-redis:development - mesh-runtime-baserow:development - mesh-runtime-letta:development diff --git a/test/integration/assigned-two-node-db.test.ts b/test/integration/assigned-two-node-db.test.ts index 7b15385..e826863 100644 --- a/test/integration/assigned-two-node-db.test.ts +++ b/test/integration/assigned-two-node-db.test.ts @@ -196,6 +196,7 @@ before(async () => { }, { timeout: 1_800_000 }); after(async () => { + if (process.env["MESH_LAB_KEEP"]) { console.log(`MESH_LAB_KEEP set — leaving ${instanceId} standing`); return; } if (instanceId) await destroy(instanceId); await destroyAll(`${SCENARIO}-`); }, { timeout: 600_000 }); @@ -372,6 +373,11 @@ test("consumers on a joined node get their databases from the one foundation sto r.image = pinned(`mesh-runtime-${name}@sha256:${"0".repeat(64)}`); delete r.artifact; } + // The provisioner runtime seals a consumer's credential; the mesh cannot yet deliver a seal + // key to a provider's runtime, so set it lab-locally — the same workaround the redis provider + // uses here (hq 04-ISSUES/022, the open provider-seal-key work). + const env = (r as { env?: Record }).env; + if (env && typeof env["MESH_RECEIVES"] === "string") env["MESH_SEAL_KEY"] = "lab-only-seal-key"; } const manifest = JSON.stringify(m); return { manifest, broker: manifest.includes("MESH_BROKER_FILE") }; @@ -397,10 +403,22 @@ test("consumers on a joined node get their databases from the one foundation sto // The packet filter on the control-node, so the from:mesh rule admits laptop's consumers to the // store over the overlay — the firewall half of cross-node provisioning (issue 055). await installCatalog("nftables", "anchor"); - // Adopt the foundation store as the ONE postgres, on the control-node: the postgres module - // reconciles the mesh-store the foundation raised and brings up its provisioner (mesh-postgres), - // which mints a database per consumer. There is no second postgres (ADR 0079). + // Adopt the foundation store AND broker as the ONE postgres and lavinmq, on the control-node: + // each module reconciles the container the foundation raised and brings up its provisioner + // (mesh-postgres mints a database per consumer; mesh-lavinmq a vhost per consumer). There is no + // second store or broker (ADR 0079). Adopting lavinmq also declares the broker's `listens` — so + // its amqps port opens in the firewall's forward chain, which is what lets a consumer on the + // joined node reach the bus cross-node at all. await installCatalog("postgres", "anchor"); + // The store's superuser is the foundation's, made at genesis (mesh-store's POSTGRES_PASSWORD) — + // carried into the module via `secret accept`, exactly as the genesis bootstrap does. Without it + // the module mints a random superuser that does not match the running store, and the provisioner + // cannot log in to create anyone's database (hq phase3 deliverSuperuser; ADR 0078). + const superPw = (await must("anchor", + `docker inspect mesh-store --format '{{range .Config.Env}}{{println .}}{{end}}' | sed -n 's/^POSTGRES_PASSWORD=//p'`)).trim(); + await must("anchor", `printf %s ${quote(superPw)} > /tmp/superuser && docker cp /tmp/superuser mesh-controller:/superuser`); + await mesh(`secret accept anchor postgres superuser --from /superuser`); + await installCatalog("lavinmq", "anchor"); await mesh(`push anchor`, 600_000); await settled("anchor"); @@ -472,22 +490,35 @@ test("consumers on a joined node get their databases from the one foundation sto // that letta is a second co-located postgres consumer that gets its OWN credential and reaches its // OWN database (the provider-seal-key gate) — is asserted above (the credential match) and below // (the live provisioning connect); its runtime `mesh-letta` and every other container stay strict. + // + // A provisioner runtime whose provider is cross-node legitimately restarts a few times at + // startup: it exits when the broker is not yet reachable (the overlay tunnel comes up a moment + // after the container does) and docker restarts it until it connects. That is startup churn, not + // a crash-loop — the difference is that churn STOPS. So wait for each container's restart count + // to settle (unchanged over a window) rather than forbid any rise; one that never settles inside + // the deadline is the real crash-loop, and it fails with the trajectory and its logs. const stable = expected.filter((n) => n !== "letta"); - const restarts = new Map(); for (const name of stable) { + const deadline = Date.now() + 300_000; + let prev = -1, stableSince = 0, last = "?"; + while (Date.now() < deadline) { + const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); + last = `running=${running} restarts=${count}`; + const n = Number(count); + if (running === "true" && n === prev) { + if (stableSince === 0) stableSince = Date.now(); + if (Date.now() - stableSince >= 30_000) break; // up and unchanged for 30s — settled + } else { + prev = n; stableSince = 0; + } + await new Promise((r) => setTimeout(r, 5_000)); + } const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); assert.equal(running, "true", - `${name} is not running after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); - restarts.set(name, Number(count)); - } - await new Promise((r) => setTimeout(r, 20000)); - for (const name of stable) { - const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); - assert.equal(running, "true", - `${name} fell over after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); - assert.ok(Number(count) <= (restarts.get(name) ?? 0), - `${name} is crash-looping (restart count rose ${restarts.get(name)} -> ${count}):\n` + - `${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + `${name} is not running after the push (${last}):\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + if (Date.now() >= deadline) + assert.fail(`${name} never stopped restarting within 300s (last ${last}) — a crash-loop, not startup churn:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + void count; } // ================================================================================================