two-node-db on the one-store model: store cross-node proven end-to-end #29

Merged
jschoubben merged 2 commits from multi-node/one-store-beds into main 2026-09-17 07:54:43 +00:00
2 changed files with 47 additions and 15 deletions
Showing only changes of commit 0e382f1db7 - Show all commits
+1
View File
@@ -48,6 +48,7 @@ images:
# its module's provisioner, so no separate mesh-provision-* image is listed — the runtime is the
# provisioner (ADR 0048).
- mesh-runtime-postgres:development
- mesh-runtime-lavinmq:development
- mesh-runtime-redis:development
- mesh-runtime-baserow:development
- mesh-runtime-letta:development
+46 -15
View File
@@ -196,6 +196,7 @@ before(async () => {
}, { timeout: 1_800_000 });
after(async () => {
if (process.env["MESH_LAB_KEEP"]) { console.log(`MESH_LAB_KEEP set — leaving ${instanceId} standing`); return; }
if (instanceId) await destroy(instanceId);
await destroyAll(`${SCENARIO}-`);
}, { timeout: 600_000 });
@@ -372,6 +373,11 @@ test("consumers on a joined node get their databases from the one foundation sto
r.image = pinned(`mesh-runtime-${name}@sha256:${"0".repeat(64)}`);
delete r.artifact;
}
// The provisioner runtime seals a consumer's credential; the mesh cannot yet deliver a seal
// key to a provider's runtime, so set it lab-locally — the same workaround the redis provider
// uses here (hq 04-ISSUES/022, the open provider-seal-key work).
const env = (r as { env?: Record<string, string> }).env;
if (env && typeof env["MESH_RECEIVES"] === "string") env["MESH_SEAL_KEY"] = "lab-only-seal-key";
}
const manifest = JSON.stringify(m);
return { manifest, broker: manifest.includes("MESH_BROKER_FILE") };
@@ -397,10 +403,22 @@ test("consumers on a joined node get their databases from the one foundation sto
// The packet filter on the control-node, so the from:mesh rule admits laptop's consumers to the
// store over the overlay — the firewall half of cross-node provisioning (issue 055).
await installCatalog("nftables", "anchor");
// Adopt the foundation store as the ONE postgres, on the control-node: the postgres module
// reconciles the mesh-store the foundation raised and brings up its provisioner (mesh-postgres),
// which mints a database per consumer. There is no second postgres (ADR 0079).
// Adopt the foundation store AND broker as the ONE postgres and lavinmq, on the control-node:
// each module reconciles the container the foundation raised and brings up its provisioner
// (mesh-postgres mints a database per consumer; mesh-lavinmq a vhost per consumer). There is no
// second store or broker (ADR 0079). Adopting lavinmq also declares the broker's `listens` — so
// its amqps port opens in the firewall's forward chain, which is what lets a consumer on the
// joined node reach the bus cross-node at all.
await installCatalog("postgres", "anchor");
// The store's superuser is the foundation's, made at genesis (mesh-store's POSTGRES_PASSWORD) —
// carried into the module via `secret accept`, exactly as the genesis bootstrap does. Without it
// the module mints a random superuser that does not match the running store, and the provisioner
// cannot log in to create anyone's database (hq phase3 deliverSuperuser; ADR 0078).
const superPw = (await must("anchor",
`docker inspect mesh-store --format '{{range .Config.Env}}{{println .}}{{end}}' | sed -n 's/^POSTGRES_PASSWORD=//p'`)).trim();
await must("anchor", `printf %s ${quote(superPw)} > /tmp/superuser && docker cp /tmp/superuser mesh-controller:/superuser`);
await mesh(`secret accept anchor postgres superuser --from /superuser`);
await installCatalog("lavinmq", "anchor");
await mesh(`push anchor`, 600_000);
await settled("anchor");
@@ -472,22 +490,35 @@ test("consumers on a joined node get their databases from the one foundation sto
// that letta is a second co-located postgres consumer that gets its OWN credential and reaches its
// OWN database (the provider-seal-key gate) — is asserted above (the credential match) and below
// (the live provisioning connect); its runtime `mesh-letta` and every other container stay strict.
//
// A provisioner runtime whose provider is cross-node legitimately restarts a few times at
// startup: it exits when the broker is not yet reachable (the overlay tunnel comes up a moment
// after the container does) and docker restarts it until it connects. That is startup churn, not
// a crash-loop — the difference is that churn STOPS. So wait for each container's restart count
// to settle (unchanged over a window) rather than forbid any rise; one that never settles inside
// the deadline is the real crash-loop, and it fails with the trajectory and its logs.
const stable = expected.filter((n) => n !== "letta");
const restarts = new Map<string, number>();
for (const name of stable) {
const deadline = Date.now() + 300_000;
let prev = -1, stableSince = 0, last = "?";
while (Date.now() < deadline) {
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
last = `running=${running} restarts=${count}`;
const n = Number(count);
if (running === "true" && n === prev) {
if (stableSince === 0) stableSince = Date.now();
if (Date.now() - stableSince >= 30_000) break; // up and unchanged for 30s — settled
} else {
prev = n; stableSince = 0;
}
await new Promise((r) => setTimeout(r, 5_000));
}
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
assert.equal(running, "true",
`${name} is not running after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
restarts.set(name, Number(count));
}
await new Promise((r) => setTimeout(r, 20000));
for (const name of stable) {
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
assert.equal(running, "true",
`${name} fell over after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
assert.ok(Number(count) <= (restarts.get(name) ?? 0),
`${name} is crash-looping (restart count rose ${restarts.get(name)} -> ${count}):\n` +
`${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
`${name} is not running after the push (${last}):\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
if (Date.now() >= deadline)
assert.fail(`${name} never stopped restarting within 300s (last ${last}) — a crash-loop, not startup churn:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
void count;
}
// ================================================================================================