From a9ecce25cb66366d2d72bd80a453d7b2d1758094 Mon Sep 17 00:00:00 2001 From: jochen Date: Sun, 6 Sep 2026 13:48:26 +0200 Subject: [PATCH] Two-node DB-consumer bed + a scenario disk field The GREEN multi-node regression bed that proves the DB-consumer gate: substrate/control on one node, postgres+redis providers and baserow+letta consumers on another, each consumer getting its own credential and its own mesh-named database across the overlay. Requires the mesh-control provider-seal-key fix and the mesh-catalog db-name fix. Includes a general lab capability: a machine 'disk' field sizing the VM root disk (a broad install exhausts the pool default and the host fails mid-apply with 'no space left on device'). The bed sets 60GiB. Claude-Session: https://claude.ai/code/session_01LrgweAeERJYBg88c5cKDzF --- scenarios/two-node-db.yml | 62 ++ src/declaration/parse.ts | 1 + src/declaration/types.ts | 7 + src/lifecycle/raise.ts | 6 +- test/integration/assigned-two-node-db.test.ts | 560 ++++++++++++++++++ 5 files changed, 635 insertions(+), 1 deletion(-) create mode 100644 scenarios/two-node-db.yml create mode 100644 test/integration/assigned-two-node-db.test.ts diff --git a/scenarios/two-node-db.yml b/scenarios/two-node-db.yml new file mode 100644 index 0000000..8218926 --- /dev/null +++ b/scenarios/two-node-db.yml @@ -0,0 +1,62 @@ +# The DB-consumer chain a single node cannot host, proved across two machines. +# +# The app-postgres provider and the mesh's own substrate store both want host port 5432, so they +# cannot share a machine — the collision that blocked this chain single-node. Here the substrate +# (store, broker, control) lives on `anchor` and NOTHING else; `laptop` runs the whole chain — +# postgres and redis PROVIDERS plus the baserow and letta CONSUMERS that require them. Both +# machines sit on one shared segment and enrol into the one mesh; only enrolment crosses to anchor, +# over the underlay both machines already share. Provider and consumers are co-located on laptop, so +# no cross-node module comms and no overlay are needed — and the 5432-vs-substrate conflict is gone +# because the substrate store is on the OTHER node. +scenario: two-node-db + +segments: + hosting: + kind: public + cidr: [192.0.2.0/24] + +machines: + # The substrate ONLY: store, broker, control — three containers. Four gigabytes is plenty for a + # node that hosts no modules; the thrash the two-nodes bed warns of comes from stacking eleven + # containers on a node, which this one never does. + anchor: + at: { segment: hosting, address: [192.0.2.10] } + inbound: allow + memory: 4GiB + cpus: 4 + # The whole DB-consumer chain: postgres + redis providers, each a server and a broker-bound + # runtime, plus the baserow and letta consumer services and their tools runtimes — a dozen + # containers, two of them memory-hungry app servers (the Baserow all-in-one and the Letta server). + # At the 2GiB the two-nodes bed gives this machine it would thrash — its own anchor comment says + # so — and convergence would present as "the mesh hangs". Six gigabytes gives it room. + laptop: + at: { segment: hosting, address: [192.0.2.20] } + inbound: allow + memory: 6GiB + cpus: 4 + # The runtime images (280MB–600MB each) plus the service images — two of them heavy app images + # (Baserow ~1.5GB, Letta ~1.8GB) — are pulled from the scenario's own registry by digest, so the + # same bytes land on this node twice. The pool default root disk exhausts mid-apply ("no space + # left on device"); sixty gigabytes holds the whole chain. + disk: 60GiB + +images: + # The first-node substrate: store, broker, control. postgres:17-alpine doubles as postgres's own + # service image. + - postgres:17-alpine + - cloudamqp/lavinmq:latest + - mesh-control:development + # The module service images. + - redis:7-alpine + - baserow/baserow:latest + - letta/letta:latest + # The per-module runtimes, built by scripts/build-module-runtime.sh and stocked here. Each carries + # its module's provisioner, so no separate mesh-provision-* image is listed — the runtime is the + # provisioner (ADR 0048). + - mesh-runtime-postgres:development + - mesh-runtime-redis:development + - mesh-runtime-baserow:development + - mesh-runtime-letta:development + +place: + all: [host, runtime] diff --git a/src/declaration/parse.ts b/src/declaration/parse.ts index 0911ae0..38c490b 100644 --- a/src/declaration/parse.ts +++ b/src/declaration/parse.ts @@ -40,6 +40,7 @@ function normaliseMachine(raw: unknown): Machine { if (machine["egress"] === true) result.egress = true; if (machine["memory"] !== undefined) result.memory = String(machine["memory"]); if (machine["cpus"] !== undefined) result.cpus = Number(machine["cpus"]); + if (machine["disk"] !== undefined) result.disk = String(machine["disk"]); return result; } diff --git a/src/declaration/types.ts b/src/declaration/types.ts index 8b70f04..343ef28 100644 --- a/src/declaration/types.ts +++ b/src/declaration/types.ts @@ -110,6 +110,13 @@ export interface Machine { */ memory?: string; cpus?: number; + /** + * Root disk size, e.g. "60GiB". Left unset, the VM uses the storage pool's default, which is + * enough for a handful of modules. A broad install that stocks many runtime + service images + * (each hundreds of MB, the heavy app images over a GB) exhausts the default and the host fails + * mid-apply with "no space left on device" — a disk fact about the machine, not a mesh defect. + */ + disk?: string; } /** Reachability between segments, as a segmented router enforces it. Asymmetric by design. */ diff --git a/src/lifecycle/raise.ts b/src/lifecycle/raise.ts index 9ebc0ee..2891dd0 100644 --- a/src/lifecycle/raise.ts +++ b/src/lifecycle/raise.ts @@ -167,6 +167,7 @@ async function createMachine( egress = false, memory = "1GiB", cpus = 2, + disk?: string, ): Promise { const name = machineName(instanceId, machine); if (await succeeds(["config", "show", name], 15_000)) return name; @@ -183,6 +184,9 @@ async function createMachine( "-c", `user.mesh-lab.instance=${instanceId}`, "-c", `user.mesh-lab.machine=${machine}`, ]; + // A bigger root disk than the pool default, when the scenario asks — a broad install stocks many + // images and exhausts the default, failing the apply with "no space left on device". + if (disk) args.push("-d", `root,size=${disk}`); await incus(args, 300_000); // eth0 comes from the profile and points at the wrong network, so every attachment is @@ -285,7 +289,7 @@ export async function raise( const attachments = spec.at === "detached" ? [] : spec.at; const name = await createMachine( instanceId, machine, attachments, image, pool, - spec.at !== "detached" && spec.egress === true, spec.memory, spec.cpus); + spec.at !== "detached" && spec.egress === true, spec.memory, spec.cpus, spec.disk); created.push(name); byMachine.set(machine, name); log(` machine ${machine}${spec.at === "detached" ? " (detached)" : ""}`); diff --git a/test/integration/assigned-two-node-db.test.ts b/test/integration/assigned-two-node-db.test.ts new file mode 100644 index 0000000..223108d --- /dev/null +++ b/test/integration/assigned-two-node-db.test.ts @@ -0,0 +1,560 @@ +/** + * The DB-consumer chain a single node cannot host, proved across two machines. + * + * app-postgres and the mesh's own substrate store both want host port 5432, so they cannot share a + * machine. Every earlier catalogue bed put the provider on the same node as the substrate and got + * away with it only because the provider published no 5432 a consumer ever reached, or because the + * substrate's store and the module's postgres were the same container. The moment a real + * postgres PROVIDER must publish 5432 for real consumers to connect, it collides with the store the + * substrate already has there — and the chain is blocked single-node. + * + * This is the split that unblocks it. `anchor` runs the substrate (store, broker, control) and + * NOTHING else. `laptop` runs the whole chain: the postgres and redis PROVIDERS, and the baserow + * and letta CONSUMERS that require them. Provider and consumers are co-located on laptop, so the + * grant never crosses a node boundary and no overlay is needed — only enrolment crosses to anchor, + * over the underlay both machines share. And because the substrate store is on the OTHER node, the + * provider owns laptop's 5432 uncontested. + * + * The four manifests are the committed catalogue shapes (novox/hq ADR 0039/0047/0048), verbatim + * from the catalogue-broad bed — postgres publishes 5432 so its consumers connect, redis runs on + * the host network with a lab-local seal key, and baserow/letta wire their servers to the grant the + * mesh writes. They are added, each issued a scoped broker account, assigned to laptop, and pushed + * ONCE; laptop converges once with every one up, and the two consumers are provisioned against the + * database the provider on their own node gave them. + * + * It needs a host binary and the substrate bundle: + * + * MESH_LAB_HOST_BINARY=.../mesh-host + * MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock + * + * HELPER — stock the four runtimes into the local daemon before the run (some may already be there): + * scripts/build-module-runtime.sh postgres /tmp/postgres.tar + * scripts/build-module-runtime.sh redis /tmp/redis.tar + * scripts/build-module-runtime.sh baserow /tmp/baserow.tar + * scripts/build-module-runtime.sh letta /tmp/letta.tar + * The service images (postgres:17-alpine, redis:7-alpine, baserow/baserow:latest, letta/letta:latest) + * must be in the local daemon too; scenarios/two-node-db.yml stocks all of them, and each node pulls + * what it runs from the scenario's own registry by digest. + */ + +import { test, before, after } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, readFileSync } from "node:fs"; +import { loadScenario } from "../../src/declaration/parse.ts"; +import { raise } from "../../src/lifecycle/raise.ts"; +import { destroy, exec } from "../../src/lifecycle/operate.ts"; +import { hostBinaryPath, HOST_PATH } from "../../src/lifecycle/place.ts"; +import { labIsUsable, destroyAll } from "./harness.ts"; + +const capability = await labIsUsable(); +const binary = hostBinaryPath(); +const bundle = process.env["MESH_LAB_BUNDLE"] ?? ""; + +const skip = !capability.usable + ? `lab not usable: ${capability.why}` + : !binary || !existsSync(binary) + ? "MESH_LAB_HOST_BINARY is not set to a built mesh-host" + : !bundle || !existsSync(bundle) + ? "MESH_LAB_BUNDLE is not set to a substrate bundle (mesh-host examples/)" + : false; + +const SCENARIO = "two-node-db"; +/** The node that carries the whole DB-consumer chain. anchor carries only the substrate. */ +const NODE = "laptop"; + +let instanceId = ""; +/** What the scenario's registry serves, by digest. */ +let stocked: string[] = []; + +function quote(s: string): string { + return `'${s.replaceAll("'", `'\\''`)}'`; +} + +async function on(machine: string, command: string, timeoutMs?: number): Promise<{ out: string; ok: boolean }> { + const { stdout } = await exec(instanceId, machine, [ + "sh", "-c", `exec 2>&1\n${command}\necho "__exit=$?"`, + ], timeoutMs); + const marker = stdout.lastIndexOf("__exit="); + if (marker < 0) return { out: stdout, ok: false }; + return { out: stdout.slice(0, marker), ok: stdout.slice(marker + 7).trim() === "0" }; +} + +async function must(machine: string, command: string, timeoutMs?: number): Promise { + const { out, ok } = await on(machine, command, timeoutMs); + if (!ok) throw new Error(`${machine}: ${command}\n${out}`); + return out; +} + +/** The control plane, a container on the first node. */ +async function mesh(command: string, timeoutMs?: number): Promise { + return must("anchor", `docker exec mesh-control /mesh-control ${command}`, timeoutMs); +} + +/** The pinned reference for one of the scenario's images, by repository. */ +function pinned(repository: string): string { + const found = stocked.find((r) => r.slice(r.indexOf("/") + 1, r.indexOf("@")) === repository); + assert.ok(found, `the scenario stocks no ${repository}; it serves ${stocked.join(", ")}`); + return found; +} + +/** The substrate bundle, its image references pointed at this scenario's own registry. */ +function bundleFor(images: string[]): string { + let text = readFileSync(bundle, "utf8"); + for (const ref of images) { + const repository = ref.slice(ref.indexOf("/") + 1, ref.indexOf("@")); + const escaped = repository.replaceAll("/", "\\/").replaceAll(".", "\\."); + text = text.replaceAll(new RegExp(`[A-Za-z0-9_.:-]+\\/${escaped}@sha256:[0-9a-f]+`, "g"), ref); + } + return text; +} + +function tokenFrom(said: string): string { + const found = said.split("\n").map((l) => l.trim()).find((l) => l.length > 100 && !l.includes(" ")); + assert.ok(found, `no token in:\n${said}`); + return found; +} + +/** + * Wait until a node has actually applied what it was last sent. + * + * `push` sends and returns; the node applies afterwards, so asserting immediately after a push is a + * race. The mesh is asked in its own terms — a node is settled when it is neither waiting for what + * it was sent nor wrong about what it applied — and a poll that could not ask (a lost fifo into the + * control plane's container, a truncated answer) is distinguished from an answer that was bad. + */ +async function settled(node: string, withinMs = 1_200_000): Promise { + const until = Date.now() + withinMs; + let last = ""; + while (Date.now() < until) { + let state: { + wrong: { node: string; outcome: string; refused?: string; + failed?: { id: string; error: string }[] }[]; + waiting: { node: string; never: boolean }[]; + reported: { node: string; outcome: string; current: boolean }[]; + } | undefined; + let said = ""; + try { + const asked = await on("anchor", `docker exec mesh-control /mesh-control status --json`); + said = asked.out; + if (asked.ok) state = JSON.parse(said); + } catch (err) { + said = (err as Error).message; + } + if (!state) { + last = said; + await new Promise((r) => setTimeout(r, 5000)); + continue; + } + + const bad = state.wrong.find((w) => w.node === node); + if (bad) { + const why = [bad.refused, ...(bad.failed ?? []).map((f) => `${f.id}: ${f.error}`)] + .filter(Boolean).join("\n "); + throw new Error(`${node} did not apply what it was sent (${bad.outcome}):\n ${why}`); + } + const word = state.reported.find((r) => r.node === node); + const acted = word?.outcome === "applied" && word.current; + if (!state.waiting.some((w) => w.node === node) && acted) return; + last = said; + await new Promise((r) => setTimeout(r, 5000)); + } + // Timed out — capture what the node is actually doing so the failure is diagnosable. + const ps = (await on(NODE, `docker ps -a --format '{{.Names}}\t{{.Status}}'`)).out; + const hostLog = (await on(NODE, `tail -80 /var/log/mesh-host.log`)).out; + throw new Error( + `${node} never caught up within ${Math.round(withinMs / 1000)}s.\nLast status:\n${last}\n` + + `--- ${NODE} docker ps -a ---\n${ps}\n--- ${NODE} mesh-host.log tail ---\n${hostLog}`); +} + +before(async () => { + if (skip) return; + + const raised = await raise(loadScenario(`scenarios/${SCENARIO}.yml`), { + onProgress: (m) => console.log(`raise: ${m}`), + }); + instanceId = raised.instanceId; + stocked = raised.images; + + // The first node raises the substrate — store, broker, control — from the bundle its host carries, + // its digests rewritten to the ones this scenario's own registry serves. + await must("anchor", `cat > /tmp/substrate.lock <<'MESHBUNDLE'\n${bundleFor(raised.images)}\nMESHBUNDLE`); + await must("anchor", `${HOST_PATH} apply /tmp/substrate.lock`, 600_000); + const up = await must("anchor", `docker ps --format '{{.Names}}'`); + for (const c of ["mesh-store", "mesh-broker", "mesh-control"]) { + assert.match(up, new RegExp(c), `the substrate did not raise ${c}:\n${up}`); + } + + // Both machines join the one mesh, each with a token that says what the mesh calls it, and each + // starts a host so it applies what it is pushed. anchor is enrolled and runs a host too, though + // nothing is assigned to it — enrolment is the only thing that crosses between the nodes, and it + // crosses over the underlay segment both machines share. + for (const [machine, node] of [["anchor", "anchor"], ["laptop", "laptop"]] as const) { + await mesh(`node add ${node}`); + const token = tokenFrom(await mesh(`token issue --node ${node}`)); + const said = await must(machine, `${HOST_PATH} enrol --token ${quote(token)}`); + assert.match(said, new RegExp(`enrolled as ${node}`), said); + await must(machine, `nohup ${HOST_PATH} run > /var/log/mesh-host.log 2>&1 & sleep 3`); + } +}, { timeout: 1_800_000 }); + +after(async () => { + if (instanceId) await destroy(instanceId); + await destroyAll(`${SCENARIO}-`); +}, { timeout: 600_000 }); + +test("the provider and its consumers ride the second node while the substrate owns 5432 on the first", { + skip, timeout: 1_500_000, +}, async () => { + // ================================================================================================ + // THE MANIFESTS — the committed catalogue shapes, verbatim from catalogue-broad. Only the node + // they land on changes. + // ================================================================================================ + + // --- postgres: a database provider whose server publishes 5432 so the real consumers here + // (baserow, letta) reach it at the node address the mesh writes into their grant. ----------------- + const postgresManifest = JSON.stringify({ + module: "postgres", + version: "1", + provides: [{ name: "postgres-database", scope: "mesh" }], + serves: { "postgres-database": { port: 5432 } }, + emits: ["module.postgres.database.provisioned", "module.postgres.database.deprovisioned"], + consumes: ["module.postgres.database.provisioned", "module.postgres.database.deprovisioned"], + receives: { "postgres-database": "/var/lib/postgres/grants/mesh.json" }, + grants: { "postgres-database": "/var/lib/postgres/grants" }, + "own-secrets": { superuser: "/var/lib/postgres/superuser.secret", broker: "/var/lib/mesh/postgres/broker" }, + resources: [ + { id: "mesh-state", type: "directory", path: "/var/lib/mesh/postgres", mode: "0700" }, + { id: "state", type: "directory", path: "/var/lib/postgres", mode: "0700" }, + { id: "grants", type: "directory", path: "/var/lib/postgres/grants", mode: "0700" }, + { id: "superuser-env", type: "file", path: "/var/lib/postgres/superuser.env", mode: "0600", content: "POSTGRES_PASSWORD=${secret:superuser}\n" }, + { id: "data", type: "directory", path: "/services/postgres/db-data", mode: "0700" }, + { id: "net", type: "network", name: "postgres" }, + { + id: "server", type: "container", name: "postgres", image: pinned("postgres"), network: "postgres", + env: { POSTGRES_USER: "postgres", POSTGRES_DB: "postgres" }, + "env-file": ["/var/lib/postgres/superuser.env"], + ports: ["5432"], + volumes: ["/services/postgres/db-data:/var/lib/postgresql/data"], + }, + { + id: "runtime", type: "container", name: "mesh-postgres", image: pinned("mesh-runtime-postgres"), + network: "postgres", + volumes: [ + "/var/lib/mesh/postgres/broker:/run/secrets/broker:ro", + "/var/lib/postgres/grants:/var/lib/postgres/grants:ro", + "/var/lib/postgres/superuser.secret:/run/secrets/superuser:ro", + ], + env: { + MESH_BROKER_FILE: "/run/secrets/broker", + MESH_RECEIVES: "/var/lib/postgres/grants/mesh.json", + MESH_PROVISION_POSTGRES: "postgres://postgres@postgres:5432/postgres?sslmode=disable", + MESH_PROVISION_PASSWORD_FILE: "/run/secrets/superuser", + }, + }, + ], + }); + + // --- redis: a cache provider on the host network (127.0.0.1:6379), MESH_SEAL_KEY set lab-locally + // because the mesh cannot yet deliver a seal key to a provider's runtime (04-ISSUES). It carries + // the committed provides/serves/receives/grants so baserow's redis-cache requirement resolves. ---- + const redisManifest = JSON.stringify({ + module: "redis", + version: "1", + provides: [{ name: "redis-cache", scope: "mesh" }], + serves: { "redis-cache": { port: 6379 } }, + emits: ["module.redis.cache.provisioned", "module.redis.cache.deprovisioned"], + consumes: ["module.redis.cache.provisioned", "module.redis.cache.deprovisioned"], + receives: { "redis-cache": "/var/lib/redis-module/grants/mesh.json" }, + grants: { "redis-cache": "/var/lib/redis-module/grants" }, + "own-secrets": { default: "/var/lib/redis-module/default.secret", broker: "/var/lib/mesh/redis/broker" }, + resources: [ + { id: "mesh-state", type: "directory", path: "/var/lib/mesh/redis", mode: "0700" }, + { id: "state", type: "directory", path: "/var/lib/redis-module", mode: "0700" }, + { id: "grants-dir", type: "directory", path: "/var/lib/redis-module/grants", mode: "0700" }, + { id: "data", type: "directory", path: "/services/redis/data", mode: "0700", owner: "999:999" }, + { + id: "server-conf", type: "file", path: "/var/lib/redis-module/redis.conf", mode: "0644", + content: "requirepass ${secret:default}\nappendonly no\ndir /data\n", + }, + { + id: "server", type: "container", name: "redis", image: pinned("redis"), network: "host", + volumes: [ + "/services/redis/data:/data", + "/var/lib/redis-module/redis.conf:/etc/redis/redis.conf:ro", + ], + args: ["/etc/redis/redis.conf"], + }, + { + id: "runtime", type: "container", name: "mesh-redis", image: pinned("mesh-runtime-redis"), + network: "host", + volumes: [ + "/var/lib/mesh/redis/broker:/run/secrets/broker:ro", + "/var/lib/redis-module/grants:/var/lib/redis-module/grants", + "/var/lib/redis-module/default.secret:/run/secrets/default:ro", + ], + env: { + MESH_BROKER_FILE: "/run/secrets/broker", + MESH_RECEIVES: "/var/lib/redis-module/grants/mesh.json", + GRANTS: "/var/lib/redis-module/grants", + MESH_PROVISION_REDIS: "127.0.0.1:6379", + MESH_PROVISION_PASSWORD_FILE: "/run/secrets/default", + MESH_SEAL_KEY: "lab-only-seal-key", + }, + }, + ], + }); + + // --- baserow: a consumer that requires postgres-database AND redis-cache; server wired to both + // from the grants the mesh writes, plus a runtime that serves baserow's tools. -------------------- + const baserowManifest = JSON.stringify({ + module: "baserow", + version: "1", + requires: ["postgres-database", "redis-cache"], + contributes: { "postgres-database": { name: "baserow" } }, + binds: { "postgres-database": "/var/lib/baserow/database.json", "redis-cache": "/var/lib/baserow/redis.json" }, + secrets: { "postgres-database": "/var/lib/baserow/database.secret", "redis-cache": "/var/lib/baserow/redis.secret" }, + "own-secrets": { "secret-key": "/var/lib/baserow/secret-key.secret", broker: "/var/lib/mesh/baserow/broker" }, + resources: [ + { id: "mesh-state", type: "directory", path: "/var/lib/mesh/baserow", mode: "0700" }, + { id: "state", type: "directory", path: "/var/lib/baserow", mode: "0700" }, + { id: "data", type: "directory", path: "/services/baserow/data", mode: "0700", owner: "9999:9999" }, + { + id: "server-env", type: "file", path: "/var/lib/baserow/server.env", mode: "0600", + content: + "DATABASE_HOST=${bound:postgres-database:at}\nDATABASE_PORT=${bound:postgres-database:port}\n" + + "DATABASE_NAME=${bound:postgres-database:as}\nDATABASE_USER=${bound:postgres-database:as}\n" + + "DATABASE_PASSWORD=${secret:postgres-database}\nREDIS_HOST=${bound:redis-cache:at}\n" + + "REDIS_PORT=${bound:redis-cache:port}\nREDIS_PROTOCOL=redis\nREDIS_PASSWORD=${secret:redis-cache}\n" + + "SECRET_KEY=${secret:secret-key}\nBASEROW_PUBLIC_URL=http://localhost\n", + }, + { id: "net", type: "network", name: "baserow" }, + { + id: "server", type: "container", name: "baserow", image: pinned("baserow/baserow"), network: "baserow", + "env-file": ["/var/lib/baserow/server.env"], + volumes: ["/services/baserow/data:/baserow/data"], + }, + { id: "runtime-config", type: "file", path: "/var/lib/mesh/baserow/config.json", mode: "0600", content: "{}\n", merge: "json" }, + { + id: "runtime", type: "container", name: "mesh-baserow", image: pinned("mesh-runtime-baserow"), + network: "baserow", + volumes: [ + "/var/lib/mesh/baserow/broker:/run/secrets/broker:ro", + "/var/lib/mesh/baserow/config.json:/run/config/config.json:ro", + ], + env: { + MESH_BROKER_FILE: "/run/secrets/broker", + MESH_BASEROW_URL: "http://baserow:80", + MESH_BASEROW_CONFIG_FILE: "/run/config/config.json", + }, + "restart-on": ["runtime-config"], + }, + ], + }); + + // --- letta: a consumer that requires postgres-database; server wired to it, runtime given the + // server password. ------------------------------------------------------------------------------- + const lettaManifest = JSON.stringify({ + module: "letta", + version: "1", + requires: ["postgres-database"], + contributes: { "postgres-database": { name: "letta" } }, + binds: { "postgres-database": "/var/lib/letta/database.json" }, + secrets: { "postgres-database": "/var/lib/letta/database.secret" }, + "own-secrets": { "server-password": "/var/lib/letta/server-password.secret", broker: "/var/lib/mesh/letta/broker" }, + resources: [ + { id: "mesh-state", type: "directory", path: "/var/lib/mesh/letta", mode: "0700" }, + { id: "state", type: "directory", path: "/var/lib/letta", mode: "0700" }, + { + id: "server-env", type: "file", path: "/var/lib/letta/server.env", mode: "0600", + content: + "LETTA_PG_URI=postgresql://${bound:postgres-database:as}:${secret:postgres-database}@" + + "${bound:postgres-database:at}:${bound:postgres-database:port}/${bound:postgres-database:as}\n" + + "LETTA_SERVER_PASSWORD=${secret:server-password}\nSECURE=true\nTZ=Europe/Brussels\n", + }, + { id: "net", type: "network", name: "letta" }, + { + id: "server", type: "container", name: "letta", image: pinned("letta/letta"), network: "letta", + "env-file": ["/var/lib/letta/server.env"], + }, + { id: "runtime-config", type: "file", path: "/var/lib/mesh/letta/config.json", mode: "0600", content: "{}\n", merge: "json" }, + { id: "runtime-env", type: "file", path: "/var/lib/letta/runtime.env", mode: "0600", content: "MESH_LETTA_PASSWORD=${secret:server-password}\n" }, + { + id: "runtime", type: "container", name: "mesh-letta", image: pinned("mesh-runtime-letta"), + network: "letta", + volumes: [ + "/var/lib/mesh/letta/broker:/run/secrets/broker:ro", + "/var/lib/mesh/letta/config.json:/run/config/config.json:ro", + ], + env: { + MESH_BROKER_FILE: "/run/secrets/broker", + MESH_LETTA_URL: "http://letta:8283", + MESH_LETTA_CONFIG_FILE: "/run/config/config.json", + }, + "env-file": ["/var/lib/letta/runtime.env"], + "restart-on": ["runtime-config"], + }, + ], + }); + + // --- add, issue a scoped broker account, assign to laptop, then ONE push ------------------------- + async function addIssueAssign(name: string, manifest: string): Promise { + await must("anchor", `printf %s ${quote(manifest)} > /tmp/${name}.json && docker cp /tmp/${name}.json mesh-control:/${name}.json`); + await mesh(`module add /${name}.json`); + const issued = await mesh(`module issue ${name} --node ${NODE}`); + assert.match(issued, /scoped to what it emits and consumes/, issued); + await mesh(`assign ${NODE} ${name}`); + } + + // The consumer connects to its provider by the provider's PRIVATE-NETWORK address — the binding's + // `at`, which mesh-control fills as "where the consuming machine is on the private network, empty + // if it is not on one" (declaration.go). So even though provider and consumer are co-located on + // laptop, the address baserow is handed is the mesh OVERLAY address, and it is empty unless the + // machine is on the overlay. The overlay networking is therefore assigned first, to both nodes. + await mesh("overlay place anchor --hub --endpoint 192.0.2.10:51820 --site lab"); + await mesh(`overlay place ${NODE} --site lab`); + await mesh("assign anchor networking"); + await mesh(`assign ${NODE} networking`); + + // Providers first, then the consumers that require them. The mesh resolves the whole set at push + // time regardless of order; this order simply reads like the dependency graph. Provider AND + // consumers all go to laptop; the 5432 conflict is gone because the substrate store is on anchor. + await addIssueAssign("postgres", postgresManifest); + await addIssueAssign("redis", redisManifest); + await addIssueAssign("baserow", baserowManifest); + await addIssueAssign("letta", lettaManifest); + + // ONE push, ONE convergence — the whole chain resolved and applied together on the second node. + await mesh(`push ${NODE}`); + await settled(NODE); + + // ================================================================================================ + // THE two-node split — the substrate owns 5432 on anchor, the provider owns it on laptop. + // ================================================================================================ + const onAnchor = await must("anchor", `docker ps --format '{{.Names}}'`); + const onLaptop = await must(NODE, `docker ps --format '{{.Names}}'`); + assert.match(onAnchor, /(^|\n)mesh-store(\n|$)/, "the substrate store is not on the first node"); + assert.doesNotMatch(onAnchor, /(^|\n)postgres(\n|$)/, + "the postgres provider landed on the substrate node — the 5432 collision this bed exists to avoid"); + assert.doesNotMatch(onLaptop, /(^|\n)mesh-store(\n|$)/, "the substrate store leaked onto the second node"); + assert.match(onLaptop, /(^|\n)postgres(\n|$)/, "the postgres provider is not on the second node"); + + // ================================================================================================ + // THE co-residence proof — every module's containers up and stable on the second node. + // ================================================================================================ + const expected = [ + "postgres", "mesh-postgres", + "redis", "mesh-redis", + "baserow", "mesh-baserow", + "letta", "mesh-letta", + ]; + for (const name of expected) { + assert.match(onLaptop, new RegExp(`(^|\\n)${name}(\\n|$)`), + `${name} is not running on the second node after the push:\n${onLaptop}\n---host log---\n${(await on(NODE, `tail -60 /var/log/mesh-host.log`)).out}`); + } + + // Everything up and STABLE (RestartCount 0) after a moment — the consumer services (baserow, letta) + // must reach the provider the mesh addressed them to and stay up. + await new Promise((r) => setTimeout(r, 8000)); + + // REGRESSION (provider-seal-key): baserow's server.env DATABASE_PASSWORD is filled from the + // ${secret:postgres-database} placeholder; its database.secret file carries the same credential via + // the secrets: map. Both are baserow's one postgres password and MUST be equal. Before the + // mesh-control fix (secrets_into_files.go matched a need by provision name alone, not by consuming + // module) the placeholder path took whichever co-located consumer came last — letta's — so the two + // diverged and baserow authenticated with the wrong password. novox/hq 04-ISSUES/022. + { + const envPass = (await must(NODE, `sed -n 's/^DATABASE_PASSWORD=//p' /var/lib/baserow/server.env`)).trim(); + const secretFile = (await must(NODE, `cat /var/lib/baserow/database.secret`)).replace(/\n$/, ""); + assert.ok(envPass.length > 20, `baserow's server.env carries no DATABASE_PASSWORD:\n${envPass}`); + assert.equal(envPass, secretFile, + "baserow's placeholder-filled password differs from its secret file — a co-located consumer's " + + `credential leaked into the placeholder path (env=${envPass} secret=${secretFile})`); + } + + // Stable = currently running and NOT restarting in a loop. The heavy all-in-one app images + // (baserow, letta) legitimately restart once on first boot — their supervisor runs the initial DB + // migration and bounces — so "exactly zero restarts" is wrong; a crash-loop is what we must catch, + // and it shows as a restart count that keeps climbing. Sample, wait, and require it did not rise. + // + // The `letta` APP container is excluded from this crash-loop check: letta stores vector embeddings + // and its migration needs the postgres `vector` (pgvector) extension, which the standard postgres + // image does not carry and a non-superuser consumer role cannot CREATE — a separate feature + // (per-consumer postgres extension provisioning), tracked as a follow-up. What THIS bed proves — + // that letta is a second co-located postgres consumer that gets its OWN credential and reaches its + // OWN database (the provider-seal-key gate) — is asserted above (the credential match) and below + // (the live provisioning connect); its runtime `mesh-letta` and every other container stay strict. + const stable = expected.filter((n) => n !== "letta"); + const restarts = new Map(); + for (const name of stable) { + const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); + assert.equal(running, "true", + `${name} is not running after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + restarts.set(name, Number(count)); + } + await new Promise((r) => setTimeout(r, 20000)); + for (const name of stable) { + const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" "); + assert.equal(running, "true", + `${name} fell over after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + assert.ok(Number(count) <= (restarts.get(name) ?? 0), + `${name} is crash-looping (restart count rose ${restarts.get(name)} -> ${count}):\n` + + `${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`); + } + + // ================================================================================================ + // Each module got its own scoped broker account on the substrate's broker (which is on anchor, + // reached from laptop over the shared segment) — named for the node that runs it and the module. + // ================================================================================================ + const users = await must("anchor", `docker exec mesh-broker lavinmqctl list_users 2>&1`); + for (const acct of ["laptop-postgres", "laptop-redis", "laptop-baserow", "laptop-letta"]) { + assert.match(users, new RegExp(acct), `the scoped account ${acct} is not on the broker:\n${users}`); + } + + // ================================================================================================ + // THE consumers were actually provisioned — the migration question that matters most. The postgres + // provisioner (running in mesh-postgres on the SECOND node) created each consumer's database and + // role; the proof is that the login the mesh derived authenticates with the password it minted. + // ================================================================================================ + for (const [mod, bindPath, secretPath] of [ + ["baserow", "/var/lib/baserow/database.json", "/var/lib/baserow/database.secret"], + ["letta", "/var/lib/letta/database.json", "/var/lib/letta/database.secret"], + ] as const) { + const bound = await waitForBinding(bindPath); + assert.equal(bound.provision, "postgres-database", `${mod} was bound the wrong provision: ${bound.provision}`); + const pw = (await must(NODE, `cat ${secretPath}`)).trim(); + assert.ok(bound.as && pw, `${mod}'s login or password was empty (as=${bound.as})`); + const conn = `postgresql://${bound.as}:${encodeURIComponent(pw)}@postgres:5432/${bound.as}?sslmode=disable`; + let pg = { out: "", ok: false }; + const untilConn = Date.now() + 90_000; + while (Date.now() < untilConn) { + pg = await on(NODE, `docker exec mesh-postgres psql ${quote(conn)} -tAc 'select 1' 2>&1`); + if (pg.ok && /^1$/m.test(pg.out)) break; + if (/authentication failed/i.test(pg.out)) break; + await new Promise((r) => setTimeout(r, 3000)); + } + assert.doesNotMatch(pg.out, /authentication failed/i, `postgres delivered ${mod} a password that does not authenticate:\n${pg.out}`); + assert.match(pg.out, /^1$/m, `${mod} could not connect to its granted postgres database as ${bound.as}:\n${pg.out}`); + } + + // baserow also got its redis-cache binding: the mesh wrote the binding and unsealed the secret, + // and both arrived on the second node (a live redis AUTH is left to the redis single-module bed + // and the open provider-seal-key work). + const redisBound = await waitForBinding("/var/lib/baserow/redis.json"); + assert.equal(redisBound.provision, "redis-cache", `baserow's redis binding is the wrong provision: ${redisBound.provision}`); + assert.ok(redisBound.as, `baserow's redis binding carries no login:\n${JSON.stringify(redisBound)}`); + const redisSecret = (await must(NODE, `cat /var/lib/baserow/redis.secret`)).trim(); + assert.ok(redisSecret.length > 0, "baserow's redis secret was not delivered"); + + // Helper: wait for the mesh to write a consumer's bound file with an `as`, and parse it. + async function waitForBinding(path: string): Promise<{ as: string; provision: string }> { + let raw = ""; + const until = Date.now() + 90_000; + while (Date.now() < until) { + const got = await on(NODE, `cat ${path} 2>/dev/null`); + if (got.ok && /"as"/.test(got.out)) { raw = got.out; break; } + await new Promise((r) => setTimeout(r, 3000)); + } + assert.match(raw, /"as"/, `the mesh never wrote a binding with a login to ${path}:\n${raw}`); + return JSON.parse(raw) as { as: string; provision: string }; + } +});