two-node-db is GREEN on the one-store model — store cross-node proven end-to-end
The complete recipe, found across five runs: adopt BOTH the store (postgres) and broker (lavinmq) on the control-node — the broker's `listens` is what opens 5671 in the firewall for cross-node bus access; deliver the store's genesis superuser via `secret accept` (else the module mints a random one that cannot log in to the running store); inject the provider seal key (open hq issue 022 workaround); push the provider node again after the remote consumers (issue 057); and tolerate provisioner-runtime startup churn — a cross-node provisioner exits until the overlay tunnel is up, then settles. baserow and letta on the joined node get their databases from the one foundation store over the overlay. https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx
This commit is contained in:
@@ -196,6 +196,7 @@ before(async () => {
|
||||
}, { timeout: 1_800_000 });
|
||||
|
||||
after(async () => {
|
||||
if (process.env["MESH_LAB_KEEP"]) { console.log(`MESH_LAB_KEEP set — leaving ${instanceId} standing`); return; }
|
||||
if (instanceId) await destroy(instanceId);
|
||||
await destroyAll(`${SCENARIO}-`);
|
||||
}, { timeout: 600_000 });
|
||||
@@ -372,6 +373,11 @@ test("consumers on a joined node get their databases from the one foundation sto
|
||||
r.image = pinned(`mesh-runtime-${name}@sha256:${"0".repeat(64)}`);
|
||||
delete r.artifact;
|
||||
}
|
||||
// The provisioner runtime seals a consumer's credential; the mesh cannot yet deliver a seal
|
||||
// key to a provider's runtime, so set it lab-locally — the same workaround the redis provider
|
||||
// uses here (hq 04-ISSUES/022, the open provider-seal-key work).
|
||||
const env = (r as { env?: Record<string, string> }).env;
|
||||
if (env && typeof env["MESH_RECEIVES"] === "string") env["MESH_SEAL_KEY"] = "lab-only-seal-key";
|
||||
}
|
||||
const manifest = JSON.stringify(m);
|
||||
return { manifest, broker: manifest.includes("MESH_BROKER_FILE") };
|
||||
@@ -397,10 +403,22 @@ test("consumers on a joined node get their databases from the one foundation sto
|
||||
// The packet filter on the control-node, so the from:mesh rule admits laptop's consumers to the
|
||||
// store over the overlay — the firewall half of cross-node provisioning (issue 055).
|
||||
await installCatalog("nftables", "anchor");
|
||||
// Adopt the foundation store as the ONE postgres, on the control-node: the postgres module
|
||||
// reconciles the mesh-store the foundation raised and brings up its provisioner (mesh-postgres),
|
||||
// which mints a database per consumer. There is no second postgres (ADR 0079).
|
||||
// Adopt the foundation store AND broker as the ONE postgres and lavinmq, on the control-node:
|
||||
// each module reconciles the container the foundation raised and brings up its provisioner
|
||||
// (mesh-postgres mints a database per consumer; mesh-lavinmq a vhost per consumer). There is no
|
||||
// second store or broker (ADR 0079). Adopting lavinmq also declares the broker's `listens` — so
|
||||
// its amqps port opens in the firewall's forward chain, which is what lets a consumer on the
|
||||
// joined node reach the bus cross-node at all.
|
||||
await installCatalog("postgres", "anchor");
|
||||
// The store's superuser is the foundation's, made at genesis (mesh-store's POSTGRES_PASSWORD) —
|
||||
// carried into the module via `secret accept`, exactly as the genesis bootstrap does. Without it
|
||||
// the module mints a random superuser that does not match the running store, and the provisioner
|
||||
// cannot log in to create anyone's database (hq phase3 deliverSuperuser; ADR 0078).
|
||||
const superPw = (await must("anchor",
|
||||
`docker inspect mesh-store --format '{{range .Config.Env}}{{println .}}{{end}}' | sed -n 's/^POSTGRES_PASSWORD=//p'`)).trim();
|
||||
await must("anchor", `printf %s ${quote(superPw)} > /tmp/superuser && docker cp /tmp/superuser mesh-controller:/superuser`);
|
||||
await mesh(`secret accept anchor postgres superuser --from /superuser`);
|
||||
await installCatalog("lavinmq", "anchor");
|
||||
await mesh(`push anchor`, 600_000);
|
||||
await settled("anchor");
|
||||
|
||||
@@ -472,22 +490,35 @@ test("consumers on a joined node get their databases from the one foundation sto
|
||||
// that letta is a second co-located postgres consumer that gets its OWN credential and reaches its
|
||||
// OWN database (the provider-seal-key gate) — is asserted above (the credential match) and below
|
||||
// (the live provisioning connect); its runtime `mesh-letta` and every other container stay strict.
|
||||
//
|
||||
// A provisioner runtime whose provider is cross-node legitimately restarts a few times at
|
||||
// startup: it exits when the broker is not yet reachable (the overlay tunnel comes up a moment
|
||||
// after the container does) and docker restarts it until it connects. That is startup churn, not
|
||||
// a crash-loop — the difference is that churn STOPS. So wait for each container's restart count
|
||||
// to settle (unchanged over a window) rather than forbid any rise; one that never settles inside
|
||||
// the deadline is the real crash-loop, and it fails with the trajectory and its logs.
|
||||
const stable = expected.filter((n) => n !== "letta");
|
||||
const restarts = new Map<string, number>();
|
||||
for (const name of stable) {
|
||||
const deadline = Date.now() + 300_000;
|
||||
let prev = -1, stableSince = 0, last = "?";
|
||||
while (Date.now() < deadline) {
|
||||
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
|
||||
last = `running=${running} restarts=${count}`;
|
||||
const n = Number(count);
|
||||
if (running === "true" && n === prev) {
|
||||
if (stableSince === 0) stableSince = Date.now();
|
||||
if (Date.now() - stableSince >= 30_000) break; // up and unchanged for 30s — settled
|
||||
} else {
|
||||
prev = n; stableSince = 0;
|
||||
}
|
||||
await new Promise((r) => setTimeout(r, 5_000));
|
||||
}
|
||||
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
|
||||
assert.equal(running, "true",
|
||||
`${name} is not running after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
|
||||
restarts.set(name, Number(count));
|
||||
}
|
||||
await new Promise((r) => setTimeout(r, 20000));
|
||||
for (const name of stable) {
|
||||
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
|
||||
assert.equal(running, "true",
|
||||
`${name} fell over after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
|
||||
assert.ok(Number(count) <= (restarts.get(name) ?? 0),
|
||||
`${name} is crash-looping (restart count rose ${restarts.get(name)} -> ${count}):\n` +
|
||||
`${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
|
||||
`${name} is not running after the push (${last}):\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
|
||||
if (Date.now() >= deadline)
|
||||
assert.fail(`${name} never stopped restarting within 300s (last ${last}) — a crash-loop, not startup churn:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
|
||||
void count;
|
||||
}
|
||||
|
||||
// ================================================================================================
|
||||
|
||||
Reference in New Issue
Block a user