diff --git a/test/integration/built-store-cross-node.test.ts b/test/integration/built-store-cross-node.test.ts index 9f6cac7..b89a9e4 100644 --- a/test/integration/built-store-cross-node.test.ts +++ b/test/integration/built-store-cross-node.test.ts @@ -281,8 +281,10 @@ test("a joined node's consumers open the store and broker the mesh built and ado // open — this bed is their honest gate, red until they are fixed. await buildAndAssign("amqp-ping", NODE); - // 057: the provider node is composed again so its provisioners learn of the remote consumers. - await mesh(`push ${CONTROL}`, 600_000); + // No second push of the provider node. Issue 057's fix makes one push sufficient: pushing the + // consumer's node cascades to the machines whose declaration changed because of it — the + // provider learns of the remote consumer from that same act. This bed is the enforcement: put + // the workaround push back and the cascade is untested again. // THE BROKER, cross-node: amqp-ping's binding names the control-node's overlay name, its vhost is // minted on the one broker, and it holds the connection. @@ -319,6 +321,37 @@ test("a joined node's consumers open the store and broker the mesh built and ado await new Promise((r) => setTimeout(r, 15_000)); assert.match((await on(NODE, `docker ps --format '{{.Names}}\t{{.Status}}'`)).out, /amqp-ping\tUp/, `amqp-ping did not stay up on ${NODE}`); + // And was never restarted by the container runtime at all (hq issue 058): a runtime whose + // broker is not reachable yet waits for it in-process, so "the overlay came up a moment after + // the container" must produce zero restarts — churn here is the crash-loop 058 retired. + const restarts = (await on(NODE, `docker inspect -f '{{.RestartCount}}' amqp-ping`)).out.trim(); + assert.equal(restarts, "0", + `amqp-ping was restarted ${restarts} time(s) by the container runtime on ${NODE}; ` + + `a runtime waits for its broker in-process (hq issue 058):\n` + + (await on(NODE, `docker logs amqp-ping 2>&1 | head -20`)).out); + + // The broker survives a restart as a reachable thing, not just a running one (hq issue 063). + // The foundation's ports are opened from anywhere on input, but the broker is a published + // container port — so a cross-node dial is forwarded, and until 063 the forward chain carried + // no foundation rule: the joined node reached the broker only until its first connection's + // conntrack entry dropped. Restarting the broker here drops it deliberately; the node must + // reconnect and the vhost the provisioner holds must still be listed, proving the forward rule + // and not a surviving conntrack entry is what carries the connection. + await must(CONTROL, `docker restart mesh-broker`, 120_000); + { + const deadline = Date.now() + 180_000; + let reachable = ""; + while (Date.now() < deadline) { + reachable = (await on(NODE, + `timeout 5 bash -c 'cat < /dev/null > /dev/tcp/${ANCHOR}/5671' 2>&1 && echo REACHED`)).out; + if (/REACHED/.test(reachable)) break; + await new Promise((r) => setTimeout(r, 5_000)); + } + assert.match(reachable, /REACHED/, + `after the broker restarted, ${NODE} cannot reach it at ${ANCHOR}:5671 — the foundation's ` + + `forward rule is missing (hq issue 063):\n` + + (await on(CONTROL, `nft list ruleset 2>/dev/null | sed -n '/chain forward/,/}/p'`)).out); + } console.log(`amqp: ${amqpBound.trim().slice(0, 160)}`); });