The bed enforces one-push provisioning, a patient runtime, and a broker restart (057/058/063) #35

Merged
jschoubben merged 2 commits from bed/one-push-and-a-patient-runtime into main 2026-09-20 10:53:19 +00:00
@@ -281,8 +281,10 @@ test("a joined node's consumers open the store and broker the mesh built and ado
// open — this bed is their honest gate, red until they are fixed. // open — this bed is their honest gate, red until they are fixed.
await buildAndAssign("amqp-ping", NODE); await buildAndAssign("amqp-ping", NODE);
// 057: the provider node is composed again so its provisioners learn of the remote consumers. // No second push of the provider node. Issue 057's fix makes one push sufficient: pushing the
await mesh(`push ${CONTROL}`, 600_000); // consumer's node cascades to the machines whose declaration changed because of it — the
// provider learns of the remote consumer from that same act. This bed is the enforcement: put
// the workaround push back and the cascade is untested again.
// THE BROKER, cross-node: amqp-ping's binding names the control-node's overlay name, its vhost is // THE BROKER, cross-node: amqp-ping's binding names the control-node's overlay name, its vhost is
// minted on the one broker, and it holds the connection. // minted on the one broker, and it holds the connection.
@@ -319,6 +321,37 @@ test("a joined node's consumers open the store and broker the mesh built and ado
await new Promise((r) => setTimeout(r, 15_000)); await new Promise((r) => setTimeout(r, 15_000));
assert.match((await on(NODE, `docker ps --format '{{.Names}}\t{{.Status}}'`)).out, assert.match((await on(NODE, `docker ps --format '{{.Names}}\t{{.Status}}'`)).out,
/amqp-ping\tUp/, `amqp-ping did not stay up on ${NODE}`); /amqp-ping\tUp/, `amqp-ping did not stay up on ${NODE}`);
// And was never restarted by the container runtime at all (hq issue 058): a runtime whose
// broker is not reachable yet waits for it in-process, so "the overlay came up a moment after
// the container" must produce zero restarts — churn here is the crash-loop 058 retired.
const restarts = (await on(NODE, `docker inspect -f '{{.RestartCount}}' amqp-ping`)).out.trim();
assert.equal(restarts, "0",
`amqp-ping was restarted ${restarts} time(s) by the container runtime on ${NODE}; ` +
`a runtime waits for its broker in-process (hq issue 058):\n` +
(await on(NODE, `docker logs amqp-ping 2>&1 | head -20`)).out);
// The broker survives a restart as a reachable thing, not just a running one (hq issue 063).
// The foundation's ports are opened from anywhere on input, but the broker is a published
// container port — so a cross-node dial is forwarded, and until 063 the forward chain carried
// no foundation rule: the joined node reached the broker only until its first connection's
// conntrack entry dropped. Restarting the broker here drops it deliberately; the node must
// reconnect and the vhost the provisioner holds must still be listed, proving the forward rule
// and not a surviving conntrack entry is what carries the connection.
await must(CONTROL, `docker restart mesh-broker`, 120_000);
{
const deadline = Date.now() + 180_000;
let reachable = "";
while (Date.now() < deadline) {
reachable = (await on(NODE,
`timeout 5 bash -c 'cat < /dev/null > /dev/tcp/${ANCHOR}/5671' 2>&1 && echo REACHED`)).out;
if (/REACHED/.test(reachable)) break;
await new Promise((r) => setTimeout(r, 5_000));
}
assert.match(reachable, /REACHED/,
`after the broker restarted, ${NODE} cannot reach it at ${ANCHOR}:5671 — the foundation's ` +
`forward rule is missing (hq issue 063):\n` +
(await on(CONTROL, `nft list ruleset 2>/dev/null | sed -n '/chain forward/,/}/p'`)).out);
}
console.log(`amqp: ${amqpBound.trim().slice(0, 160)}`); console.log(`amqp: ${amqpBound.trim().slice(0, 160)}`);
}); });