Two-node DB-consumer bed (GREEN) + scenario disk field #10

Merged
jschoubben merged 1 commits from feat/two-node-db-consumer into main 2026-09-06 11:48:43 +00:00
5 changed files with 635 additions and 1 deletions
+62
View File
@@ -0,0 +1,62 @@
# The DB-consumer chain a single node cannot host, proved across two machines.
#
# The app-postgres provider and the mesh's own substrate store both want host port 5432, so they
# cannot share a machine — the collision that blocked this chain single-node. Here the substrate
# (store, broker, control) lives on `anchor` and NOTHING else; `laptop` runs the whole chain —
# postgres and redis PROVIDERS plus the baserow and letta CONSUMERS that require them. Both
# machines sit on one shared segment and enrol into the one mesh; only enrolment crosses to anchor,
# over the underlay both machines already share. Provider and consumers are co-located on laptop, so
# no cross-node module comms and no overlay are needed — and the 5432-vs-substrate conflict is gone
# because the substrate store is on the OTHER node.
scenario: two-node-db
segments:
hosting:
kind: public
cidr: [192.0.2.0/24]
machines:
# The substrate ONLY: store, broker, control — three containers. Four gigabytes is plenty for a
# node that hosts no modules; the thrash the two-nodes bed warns of comes from stacking eleven
# containers on a node, which this one never does.
anchor:
at: { segment: hosting, address: [192.0.2.10] }
inbound: allow
memory: 4GiB
cpus: 4
# The whole DB-consumer chain: postgres + redis providers, each a server and a broker-bound
# runtime, plus the baserow and letta consumer services and their tools runtimes — a dozen
# containers, two of them memory-hungry app servers (the Baserow all-in-one and the Letta server).
# At the 2GiB the two-nodes bed gives this machine it would thrash — its own anchor comment says
# so — and convergence would present as "the mesh hangs". Six gigabytes gives it room.
laptop:
at: { segment: hosting, address: [192.0.2.20] }
inbound: allow
memory: 6GiB
cpus: 4
# The runtime images (280MB–600MB each) plus the service images — two of them heavy app images
# (Baserow ~1.5GB, Letta ~1.8GB) — are pulled from the scenario's own registry by digest, so the
# same bytes land on this node twice. The pool default root disk exhausts mid-apply ("no space
# left on device"); sixty gigabytes holds the whole chain.
disk: 60GiB
images:
# The first-node substrate: store, broker, control. postgres:17-alpine doubles as postgres's own
# service image.
- postgres:17-alpine
- cloudamqp/lavinmq:latest
- mesh-control:development
# The module service images.
- redis:7-alpine
- baserow/baserow:latest
- letta/letta:latest
# The per-module runtimes, built by scripts/build-module-runtime.sh and stocked here. Each carries
# its module's provisioner, so no separate mesh-provision-* image is listed — the runtime is the
# provisioner (ADR 0048).
- mesh-runtime-postgres:development
- mesh-runtime-redis:development
- mesh-runtime-baserow:development
- mesh-runtime-letta:development
place:
all: [host, runtime]
+1
View File
@@ -40,6 +40,7 @@ function normaliseMachine(raw: unknown): Machine {
if (machine["egress"] === true) result.egress = true;
if (machine["memory"] !== undefined) result.memory = String(machine["memory"]);
if (machine["cpus"] !== undefined) result.cpus = Number(machine["cpus"]);
if (machine["disk"] !== undefined) result.disk = String(machine["disk"]);
return result;
}
+7
View File
@@ -110,6 +110,13 @@ export interface Machine {
*/
memory?: string;
cpus?: number;
/**
* Root disk size, e.g. "60GiB". Left unset, the VM uses the storage pool's default, which is
* enough for a handful of modules. A broad install that stocks many runtime + service images
* (each hundreds of MB, the heavy app images over a GB) exhausts the default and the host fails
* mid-apply with "no space left on device" — a disk fact about the machine, not a mesh defect.
*/
disk?: string;
}
/** Reachability between segments, as a segmented router enforces it. Asymmetric by design. */
+5 -1
View File
@@ -167,6 +167,7 @@ async function createMachine(
egress = false,
memory = "1GiB",
cpus = 2,
disk?: string,
): Promise<string> {
const name = machineName(instanceId, machine);
if (await succeeds(["config", "show", name], 15_000)) return name;
@@ -183,6 +184,9 @@ async function createMachine(
"-c", `user.mesh-lab.instance=${instanceId}`,
"-c", `user.mesh-lab.machine=${machine}`,
];
// A bigger root disk than the pool default, when the scenario asks — a broad install stocks many
// images and exhausts the default, failing the apply with "no space left on device".
if (disk) args.push("-d", `root,size=${disk}`);
await incus(args, 300_000);
// eth0 comes from the profile and points at the wrong network, so every attachment is
@@ -285,7 +289,7 @@ export async function raise(
const attachments = spec.at === "detached" ? [] : spec.at;
const name = await createMachine(
instanceId, machine, attachments, image, pool,
spec.at !== "detached" && spec.egress === true, spec.memory, spec.cpus);
spec.at !== "detached" && spec.egress === true, spec.memory, spec.cpus, spec.disk);
created.push(name);
byMachine.set(machine, name);
log(` machine ${machine}${spec.at === "detached" ? " (detached)" : ""}`);
@@ -0,0 +1,560 @@
/**
* The DB-consumer chain a single node cannot host, proved across two machines.
*
* app-postgres and the mesh's own substrate store both want host port 5432, so they cannot share a
* machine. Every earlier catalogue bed put the provider on the same node as the substrate and got
* away with it only because the provider published no 5432 a consumer ever reached, or because the
* substrate's store and the module's postgres were the same container. The moment a real
* postgres PROVIDER must publish 5432 for real consumers to connect, it collides with the store the
* substrate already has there — and the chain is blocked single-node.
*
* This is the split that unblocks it. `anchor` runs the substrate (store, broker, control) and
* NOTHING else. `laptop` runs the whole chain: the postgres and redis PROVIDERS, and the baserow
* and letta CONSUMERS that require them. Provider and consumers are co-located on laptop, so the
* grant never crosses a node boundary and no overlay is needed — only enrolment crosses to anchor,
* over the underlay both machines share. And because the substrate store is on the OTHER node, the
* provider owns laptop's 5432 uncontested.
*
* The four manifests are the committed catalogue shapes (novox/hq ADR 0039/0047/0048), verbatim
* from the catalogue-broad bed — postgres publishes 5432 so its consumers connect, redis runs on
* the host network with a lab-local seal key, and baserow/letta wire their servers to the grant the
* mesh writes. They are added, each issued a scoped broker account, assigned to laptop, and pushed
* ONCE; laptop converges once with every one up, and the two consumers are provisioned against the
* database the provider on their own node gave them.
*
* It needs a host binary and the substrate bundle:
*
* MESH_LAB_HOST_BINARY=.../mesh-host
* MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
*
* HELPER — stock the four runtimes into the local daemon before the run (some may already be there):
* scripts/build-module-runtime.sh postgres /tmp/postgres.tar
* scripts/build-module-runtime.sh redis /tmp/redis.tar
* scripts/build-module-runtime.sh baserow /tmp/baserow.tar
* scripts/build-module-runtime.sh letta /tmp/letta.tar
* The service images (postgres:17-alpine, redis:7-alpine, baserow/baserow:latest, letta/letta:latest)
* must be in the local daemon too; scenarios/two-node-db.yml stocks all of them, and each node pulls
* what it runs from the scenario's own registry by digest.
*/
import { test, before, after } from "node:test";
import assert from "node:assert/strict";
import { existsSync, readFileSync } from "node:fs";
import { loadScenario } from "../../src/declaration/parse.ts";
import { raise } from "../../src/lifecycle/raise.ts";
import { destroy, exec } from "../../src/lifecycle/operate.ts";
import { hostBinaryPath, HOST_PATH } from "../../src/lifecycle/place.ts";
import { labIsUsable, destroyAll } from "./harness.ts";
const capability = await labIsUsable();
const binary = hostBinaryPath();
const bundle = process.env["MESH_LAB_BUNDLE"] ?? "";
const skip = !capability.usable
? `lab not usable: ${capability.why}`
: !binary || !existsSync(binary)
? "MESH_LAB_HOST_BINARY is not set to a built mesh-host"
: !bundle || !existsSync(bundle)
? "MESH_LAB_BUNDLE is not set to a substrate bundle (mesh-host examples/)"
: false;
const SCENARIO = "two-node-db";
/** The node that carries the whole DB-consumer chain. anchor carries only the substrate. */
const NODE = "laptop";
let instanceId = "";
/** What the scenario's registry serves, by digest. */
let stocked: string[] = [];
function quote(s: string): string {
return `'${s.replaceAll("'", `'\\''`)}'`;
}
async function on(machine: string, command: string, timeoutMs?: number): Promise<{ out: string; ok: boolean }> {
const { stdout } = await exec(instanceId, machine, [
"sh", "-c", `exec 2>&1\n${command}\necho "__exit=$?"`,
], timeoutMs);
const marker = stdout.lastIndexOf("__exit=");
if (marker < 0) return { out: stdout, ok: false };
return { out: stdout.slice(0, marker), ok: stdout.slice(marker + 7).trim() === "0" };
}
async function must(machine: string, command: string, timeoutMs?: number): Promise<string> {
const { out, ok } = await on(machine, command, timeoutMs);
if (!ok) throw new Error(`${machine}: ${command}\n${out}`);
return out;
}
/** The control plane, a container on the first node. */
async function mesh(command: string, timeoutMs?: number): Promise<string> {
return must("anchor", `docker exec mesh-control /mesh-control ${command}`, timeoutMs);
}
/** The pinned reference for one of the scenario's images, by repository. */
function pinned(repository: string): string {
const found = stocked.find((r) => r.slice(r.indexOf("/") + 1, r.indexOf("@")) === repository);
assert.ok(found, `the scenario stocks no ${repository}; it serves ${stocked.join(", ")}`);
return found;
}
/** The substrate bundle, its image references pointed at this scenario's own registry. */
function bundleFor(images: string[]): string {
let text = readFileSync(bundle, "utf8");
for (const ref of images) {
const repository = ref.slice(ref.indexOf("/") + 1, ref.indexOf("@"));
const escaped = repository.replaceAll("/", "\\/").replaceAll(".", "\\.");
text = text.replaceAll(new RegExp(`[A-Za-z0-9_.:-]+\\/${escaped}@sha256:[0-9a-f]+`, "g"), ref);
}
return text;
}
function tokenFrom(said: string): string {
const found = said.split("\n").map((l) => l.trim()).find((l) => l.length > 100 && !l.includes(" "));
assert.ok(found, `no token in:\n${said}`);
return found;
}
/**
* Wait until a node has actually applied what it was last sent.
*
* `push` sends and returns; the node applies afterwards, so asserting immediately after a push is a
* race. The mesh is asked in its own terms — a node is settled when it is neither waiting for what
* it was sent nor wrong about what it applied — and a poll that could not ask (a lost fifo into the
* control plane's container, a truncated answer) is distinguished from an answer that was bad.
*/
async function settled(node: string, withinMs = 1_200_000): Promise<void> {
const until = Date.now() + withinMs;
let last = "";
while (Date.now() < until) {
let state: {
wrong: { node: string; outcome: string; refused?: string;
failed?: { id: string; error: string }[] }[];
waiting: { node: string; never: boolean }[];
reported: { node: string; outcome: string; current: boolean }[];
} | undefined;
let said = "";
try {
const asked = await on("anchor", `docker exec mesh-control /mesh-control status --json`);
said = asked.out;
if (asked.ok) state = JSON.parse(said);
} catch (err) {
said = (err as Error).message;
}
if (!state) {
last = said;
await new Promise((r) => setTimeout(r, 5000));
continue;
}
const bad = state.wrong.find((w) => w.node === node);
if (bad) {
const why = [bad.refused, ...(bad.failed ?? []).map((f) => `${f.id}: ${f.error}`)]
.filter(Boolean).join("\n ");
throw new Error(`${node} did not apply what it was sent (${bad.outcome}):\n ${why}`);
}
const word = state.reported.find((r) => r.node === node);
const acted = word?.outcome === "applied" && word.current;
if (!state.waiting.some((w) => w.node === node) && acted) return;
last = said;
await new Promise((r) => setTimeout(r, 5000));
}
// Timed out — capture what the node is actually doing so the failure is diagnosable.
const ps = (await on(NODE, `docker ps -a --format '{{.Names}}\t{{.Status}}'`)).out;
const hostLog = (await on(NODE, `tail -80 /var/log/mesh-host.log`)).out;
throw new Error(
`${node} never caught up within ${Math.round(withinMs / 1000)}s.\nLast status:\n${last}\n` +
`--- ${NODE} docker ps -a ---\n${ps}\n--- ${NODE} mesh-host.log tail ---\n${hostLog}`);
}
before(async () => {
if (skip) return;
const raised = await raise(loadScenario(`scenarios/${SCENARIO}.yml`), {
onProgress: (m) => console.log(`raise: ${m}`),
});
instanceId = raised.instanceId;
stocked = raised.images;
// The first node raises the substrate — store, broker, control — from the bundle its host carries,
// its digests rewritten to the ones this scenario's own registry serves.
await must("anchor", `cat > /tmp/substrate.lock <<'MESHBUNDLE'\n${bundleFor(raised.images)}\nMESHBUNDLE`);
await must("anchor", `${HOST_PATH} apply /tmp/substrate.lock`, 600_000);
const up = await must("anchor", `docker ps --format '{{.Names}}'`);
for (const c of ["mesh-store", "mesh-broker", "mesh-control"]) {
assert.match(up, new RegExp(c), `the substrate did not raise ${c}:\n${up}`);
}
// Both machines join the one mesh, each with a token that says what the mesh calls it, and each
// starts a host so it applies what it is pushed. anchor is enrolled and runs a host too, though
// nothing is assigned to it — enrolment is the only thing that crosses between the nodes, and it
// crosses over the underlay segment both machines share.
for (const [machine, node] of [["anchor", "anchor"], ["laptop", "laptop"]] as const) {
await mesh(`node add ${node}`);
const token = tokenFrom(await mesh(`token issue --node ${node}`));
const said = await must(machine, `${HOST_PATH} enrol --token ${quote(token)}`);
assert.match(said, new RegExp(`enrolled as ${node}`), said);
await must(machine, `nohup ${HOST_PATH} run > /var/log/mesh-host.log 2>&1 & sleep 3`);
}
}, { timeout: 1_800_000 });
after(async () => {
if (instanceId) await destroy(instanceId);
await destroyAll(`${SCENARIO}-`);
}, { timeout: 600_000 });
test("the provider and its consumers ride the second node while the substrate owns 5432 on the first", {
skip, timeout: 1_500_000,
}, async () => {
// ================================================================================================
// THE MANIFESTS — the committed catalogue shapes, verbatim from catalogue-broad. Only the node
// they land on changes.
// ================================================================================================
// --- postgres: a database provider whose server publishes 5432 so the real consumers here
// (baserow, letta) reach it at the node address the mesh writes into their grant. -----------------
const postgresManifest = JSON.stringify({
module: "postgres",
version: "1",
provides: [{ name: "postgres-database", scope: "mesh" }],
serves: { "postgres-database": { port: 5432 } },
emits: ["module.postgres.database.provisioned", "module.postgres.database.deprovisioned"],
consumes: ["module.postgres.database.provisioned", "module.postgres.database.deprovisioned"],
receives: { "postgres-database": "/var/lib/postgres/grants/mesh.json" },
grants: { "postgres-database": "/var/lib/postgres/grants" },
"own-secrets": { superuser: "/var/lib/postgres/superuser.secret", broker: "/var/lib/mesh/postgres/broker" },
resources: [
{ id: "mesh-state", type: "directory", path: "/var/lib/mesh/postgres", mode: "0700" },
{ id: "state", type: "directory", path: "/var/lib/postgres", mode: "0700" },
{ id: "grants", type: "directory", path: "/var/lib/postgres/grants", mode: "0700" },
{ id: "superuser-env", type: "file", path: "/var/lib/postgres/superuser.env", mode: "0600", content: "POSTGRES_PASSWORD=${secret:superuser}\n" },
{ id: "data", type: "directory", path: "/services/postgres/db-data", mode: "0700" },
{ id: "net", type: "network", name: "postgres" },
{
id: "server", type: "container", name: "postgres", image: pinned("postgres"), network: "postgres",
env: { POSTGRES_USER: "postgres", POSTGRES_DB: "postgres" },
"env-file": ["/var/lib/postgres/superuser.env"],
ports: ["5432"],
volumes: ["/services/postgres/db-data:/var/lib/postgresql/data"],
},
{
id: "runtime", type: "container", name: "mesh-postgres", image: pinned("mesh-runtime-postgres"),
network: "postgres",
volumes: [
"/var/lib/mesh/postgres/broker:/run/secrets/broker:ro",
"/var/lib/postgres/grants:/var/lib/postgres/grants:ro",
"/var/lib/postgres/superuser.secret:/run/secrets/superuser:ro",
],
env: {
MESH_BROKER_FILE: "/run/secrets/broker",
MESH_RECEIVES: "/var/lib/postgres/grants/mesh.json",
MESH_PROVISION_POSTGRES: "postgres://postgres@postgres:5432/postgres?sslmode=disable",
MESH_PROVISION_PASSWORD_FILE: "/run/secrets/superuser",
},
},
],
});
// --- redis: a cache provider on the host network (127.0.0.1:6379), MESH_SEAL_KEY set lab-locally
// because the mesh cannot yet deliver a seal key to a provider's runtime (04-ISSUES). It carries
// the committed provides/serves/receives/grants so baserow's redis-cache requirement resolves. ----
const redisManifest = JSON.stringify({
module: "redis",
version: "1",
provides: [{ name: "redis-cache", scope: "mesh" }],
serves: { "redis-cache": { port: 6379 } },
emits: ["module.redis.cache.provisioned", "module.redis.cache.deprovisioned"],
consumes: ["module.redis.cache.provisioned", "module.redis.cache.deprovisioned"],
receives: { "redis-cache": "/var/lib/redis-module/grants/mesh.json" },
grants: { "redis-cache": "/var/lib/redis-module/grants" },
"own-secrets": { default: "/var/lib/redis-module/default.secret", broker: "/var/lib/mesh/redis/broker" },
resources: [
{ id: "mesh-state", type: "directory", path: "/var/lib/mesh/redis", mode: "0700" },
{ id: "state", type: "directory", path: "/var/lib/redis-module", mode: "0700" },
{ id: "grants-dir", type: "directory", path: "/var/lib/redis-module/grants", mode: "0700" },
{ id: "data", type: "directory", path: "/services/redis/data", mode: "0700", owner: "999:999" },
{
id: "server-conf", type: "file", path: "/var/lib/redis-module/redis.conf", mode: "0644",
content: "requirepass ${secret:default}\nappendonly no\ndir /data\n",
},
{
id: "server", type: "container", name: "redis", image: pinned("redis"), network: "host",
volumes: [
"/services/redis/data:/data",
"/var/lib/redis-module/redis.conf:/etc/redis/redis.conf:ro",
],
args: ["/etc/redis/redis.conf"],
},
{
id: "runtime", type: "container", name: "mesh-redis", image: pinned("mesh-runtime-redis"),
network: "host",
volumes: [
"/var/lib/mesh/redis/broker:/run/secrets/broker:ro",
"/var/lib/redis-module/grants:/var/lib/redis-module/grants",
"/var/lib/redis-module/default.secret:/run/secrets/default:ro",
],
env: {
MESH_BROKER_FILE: "/run/secrets/broker",
MESH_RECEIVES: "/var/lib/redis-module/grants/mesh.json",
GRANTS: "/var/lib/redis-module/grants",
MESH_PROVISION_REDIS: "127.0.0.1:6379",
MESH_PROVISION_PASSWORD_FILE: "/run/secrets/default",
MESH_SEAL_KEY: "lab-only-seal-key",
},
},
],
});
// --- baserow: a consumer that requires postgres-database AND redis-cache; server wired to both
// from the grants the mesh writes, plus a runtime that serves baserow's tools. --------------------
const baserowManifest = JSON.stringify({
module: "baserow",
version: "1",
requires: ["postgres-database", "redis-cache"],
contributes: { "postgres-database": { name: "baserow" } },
binds: { "postgres-database": "/var/lib/baserow/database.json", "redis-cache": "/var/lib/baserow/redis.json" },
secrets: { "postgres-database": "/var/lib/baserow/database.secret", "redis-cache": "/var/lib/baserow/redis.secret" },
"own-secrets": { "secret-key": "/var/lib/baserow/secret-key.secret", broker: "/var/lib/mesh/baserow/broker" },
resources: [
{ id: "mesh-state", type: "directory", path: "/var/lib/mesh/baserow", mode: "0700" },
{ id: "state", type: "directory", path: "/var/lib/baserow", mode: "0700" },
{ id: "data", type: "directory", path: "/services/baserow/data", mode: "0700", owner: "9999:9999" },
{
id: "server-env", type: "file", path: "/var/lib/baserow/server.env", mode: "0600",
content:
"DATABASE_HOST=${bound:postgres-database:at}\nDATABASE_PORT=${bound:postgres-database:port}\n" +
"DATABASE_NAME=${bound:postgres-database:as}\nDATABASE_USER=${bound:postgres-database:as}\n" +
"DATABASE_PASSWORD=${secret:postgres-database}\nREDIS_HOST=${bound:redis-cache:at}\n" +
"REDIS_PORT=${bound:redis-cache:port}\nREDIS_PROTOCOL=redis\nREDIS_PASSWORD=${secret:redis-cache}\n" +
"SECRET_KEY=${secret:secret-key}\nBASEROW_PUBLIC_URL=http://localhost\n",
},
{ id: "net", type: "network", name: "baserow" },
{
id: "server", type: "container", name: "baserow", image: pinned("baserow/baserow"), network: "baserow",
"env-file": ["/var/lib/baserow/server.env"],
volumes: ["/services/baserow/data:/baserow/data"],
},
{ id: "runtime-config", type: "file", path: "/var/lib/mesh/baserow/config.json", mode: "0600", content: "{}\n", merge: "json" },
{
id: "runtime", type: "container", name: "mesh-baserow", image: pinned("mesh-runtime-baserow"),
network: "baserow",
volumes: [
"/var/lib/mesh/baserow/broker:/run/secrets/broker:ro",
"/var/lib/mesh/baserow/config.json:/run/config/config.json:ro",
],
env: {
MESH_BROKER_FILE: "/run/secrets/broker",
MESH_BASEROW_URL: "http://baserow:80",
MESH_BASEROW_CONFIG_FILE: "/run/config/config.json",
},
"restart-on": ["runtime-config"],
},
],
});
// --- letta: a consumer that requires postgres-database; server wired to it, runtime given the
// server password. -------------------------------------------------------------------------------
const lettaManifest = JSON.stringify({
module: "letta",
version: "1",
requires: ["postgres-database"],
contributes: { "postgres-database": { name: "letta" } },
binds: { "postgres-database": "/var/lib/letta/database.json" },
secrets: { "postgres-database": "/var/lib/letta/database.secret" },
"own-secrets": { "server-password": "/var/lib/letta/server-password.secret", broker: "/var/lib/mesh/letta/broker" },
resources: [
{ id: "mesh-state", type: "directory", path: "/var/lib/mesh/letta", mode: "0700" },
{ id: "state", type: "directory", path: "/var/lib/letta", mode: "0700" },
{
id: "server-env", type: "file", path: "/var/lib/letta/server.env", mode: "0600",
content:
"LETTA_PG_URI=postgresql://${bound:postgres-database:as}:${secret:postgres-database}@" +
"${bound:postgres-database:at}:${bound:postgres-database:port}/${bound:postgres-database:as}\n" +
"LETTA_SERVER_PASSWORD=${secret:server-password}\nSECURE=true\nTZ=Europe/Brussels\n",
},
{ id: "net", type: "network", name: "letta" },
{
id: "server", type: "container", name: "letta", image: pinned("letta/letta"), network: "letta",
"env-file": ["/var/lib/letta/server.env"],
},
{ id: "runtime-config", type: "file", path: "/var/lib/mesh/letta/config.json", mode: "0600", content: "{}\n", merge: "json" },
{ id: "runtime-env", type: "file", path: "/var/lib/letta/runtime.env", mode: "0600", content: "MESH_LETTA_PASSWORD=${secret:server-password}\n" },
{
id: "runtime", type: "container", name: "mesh-letta", image: pinned("mesh-runtime-letta"),
network: "letta",
volumes: [
"/var/lib/mesh/letta/broker:/run/secrets/broker:ro",
"/var/lib/mesh/letta/config.json:/run/config/config.json:ro",
],
env: {
MESH_BROKER_FILE: "/run/secrets/broker",
MESH_LETTA_URL: "http://letta:8283",
MESH_LETTA_CONFIG_FILE: "/run/config/config.json",
},
"env-file": ["/var/lib/letta/runtime.env"],
"restart-on": ["runtime-config"],
},
],
});
// --- add, issue a scoped broker account, assign to laptop, then ONE push -------------------------
async function addIssueAssign(name: string, manifest: string): Promise<void> {
await must("anchor", `printf %s ${quote(manifest)} > /tmp/${name}.json && docker cp /tmp/${name}.json mesh-control:/${name}.json`);
await mesh(`module add /${name}.json`);
const issued = await mesh(`module issue ${name} --node ${NODE}`);
assert.match(issued, /scoped to what it emits and consumes/, issued);
await mesh(`assign ${NODE} ${name}`);
}
// The consumer connects to its provider by the provider's PRIVATE-NETWORK address — the binding's
// `at`, which mesh-control fills as "where the consuming machine is on the private network, empty
// if it is not on one" (declaration.go). So even though provider and consumer are co-located on
// laptop, the address baserow is handed is the mesh OVERLAY address, and it is empty unless the
// machine is on the overlay. The overlay networking is therefore assigned first, to both nodes.
await mesh("overlay place anchor --hub --endpoint 192.0.2.10:51820 --site lab");
await mesh(`overlay place ${NODE} --site lab`);
await mesh("assign anchor networking");
await mesh(`assign ${NODE} networking`);
// Providers first, then the consumers that require them. The mesh resolves the whole set at push
// time regardless of order; this order simply reads like the dependency graph. Provider AND
// consumers all go to laptop; the 5432 conflict is gone because the substrate store is on anchor.
await addIssueAssign("postgres", postgresManifest);
await addIssueAssign("redis", redisManifest);
await addIssueAssign("baserow", baserowManifest);
await addIssueAssign("letta", lettaManifest);
// ONE push, ONE convergence — the whole chain resolved and applied together on the second node.
await mesh(`push ${NODE}`);
await settled(NODE);
// ================================================================================================
// THE two-node split — the substrate owns 5432 on anchor, the provider owns it on laptop.
// ================================================================================================
const onAnchor = await must("anchor", `docker ps --format '{{.Names}}'`);
const onLaptop = await must(NODE, `docker ps --format '{{.Names}}'`);
assert.match(onAnchor, /(^|\n)mesh-store(\n|$)/, "the substrate store is not on the first node");
assert.doesNotMatch(onAnchor, /(^|\n)postgres(\n|$)/,
"the postgres provider landed on the substrate node — the 5432 collision this bed exists to avoid");
assert.doesNotMatch(onLaptop, /(^|\n)mesh-store(\n|$)/, "the substrate store leaked onto the second node");
assert.match(onLaptop, /(^|\n)postgres(\n|$)/, "the postgres provider is not on the second node");
// ================================================================================================
// THE co-residence proof — every module's containers up and stable on the second node.
// ================================================================================================
const expected = [
"postgres", "mesh-postgres",
"redis", "mesh-redis",
"baserow", "mesh-baserow",
"letta", "mesh-letta",
];
for (const name of expected) {
assert.match(onLaptop, new RegExp(`(^|\\n)${name}(\\n|$)`),
`${name} is not running on the second node after the push:\n${onLaptop}\n---host log---\n${(await on(NODE, `tail -60 /var/log/mesh-host.log`)).out}`);
}
// Everything up and STABLE (RestartCount 0) after a moment — the consumer services (baserow, letta)
// must reach the provider the mesh addressed them to and stay up.
await new Promise((r) => setTimeout(r, 8000));
// REGRESSION (provider-seal-key): baserow's server.env DATABASE_PASSWORD is filled from the
// ${secret:postgres-database} placeholder; its database.secret file carries the same credential via
// the secrets: map. Both are baserow's one postgres password and MUST be equal. Before the
// mesh-control fix (secrets_into_files.go matched a need by provision name alone, not by consuming
// module) the placeholder path took whichever co-located consumer came last — letta's — so the two
// diverged and baserow authenticated with the wrong password. novox/hq 04-ISSUES/022.
{
const envPass = (await must(NODE, `sed -n 's/^DATABASE_PASSWORD=//p' /var/lib/baserow/server.env`)).trim();
const secretFile = (await must(NODE, `cat /var/lib/baserow/database.secret`)).replace(/\n$/, "");
assert.ok(envPass.length > 20, `baserow's server.env carries no DATABASE_PASSWORD:\n${envPass}`);
assert.equal(envPass, secretFile,
"baserow's placeholder-filled password differs from its secret file — a co-located consumer's " +
`credential leaked into the placeholder path (env=${envPass} secret=${secretFile})`);
}
// Stable = currently running and NOT restarting in a loop. The heavy all-in-one app images
// (baserow, letta) legitimately restart once on first boot — their supervisor runs the initial DB
// migration and bounces — so "exactly zero restarts" is wrong; a crash-loop is what we must catch,
// and it shows as a restart count that keeps climbing. Sample, wait, and require it did not rise.
//
// The `letta` APP container is excluded from this crash-loop check: letta stores vector embeddings
// and its migration needs the postgres `vector` (pgvector) extension, which the standard postgres
// image does not carry and a non-superuser consumer role cannot CREATE — a separate feature
// (per-consumer postgres extension provisioning), tracked as a follow-up. What THIS bed proves —
// that letta is a second co-located postgres consumer that gets its OWN credential and reaches its
// OWN database (the provider-seal-key gate) — is asserted above (the credential match) and below
// (the live provisioning connect); its runtime `mesh-letta` and every other container stay strict.
const stable = expected.filter((n) => n !== "letta");
const restarts = new Map<string, number>();
for (const name of stable) {
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
assert.equal(running, "true",
`${name} is not running after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
restarts.set(name, Number(count));
}
await new Promise((r) => setTimeout(r, 20000));
for (const name of stable) {
const [running, count] = (await must(NODE, `docker inspect -f '{{.State.Running}} {{.RestartCount}}' ${name}`)).trim().split(" ");
assert.equal(running, "true",
`${name} fell over after the push:\n${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
assert.ok(Number(count) <= (restarts.get(name) ?? 0),
`${name} is crash-looping (restart count rose ${restarts.get(name)} -> ${count}):\n` +
`${(await on(NODE, `docker logs ${name} 2>&1 | tail -40`)).out}`);
}
// ================================================================================================
// Each module got its own scoped broker account on the substrate's broker (which is on anchor,
// reached from laptop over the shared segment) — named for the node that runs it and the module.
// ================================================================================================
const users = await must("anchor", `docker exec mesh-broker lavinmqctl list_users 2>&1`);
for (const acct of ["laptop-postgres", "laptop-redis", "laptop-baserow", "laptop-letta"]) {
assert.match(users, new RegExp(acct), `the scoped account ${acct} is not on the broker:\n${users}`);
}
// ================================================================================================
// THE consumers were actually provisioned — the migration question that matters most. The postgres
// provisioner (running in mesh-postgres on the SECOND node) created each consumer's database and
// role; the proof is that the login the mesh derived authenticates with the password it minted.
// ================================================================================================
for (const [mod, bindPath, secretPath] of [
["baserow", "/var/lib/baserow/database.json", "/var/lib/baserow/database.secret"],
["letta", "/var/lib/letta/database.json", "/var/lib/letta/database.secret"],
] as const) {
const bound = await waitForBinding(bindPath);
assert.equal(bound.provision, "postgres-database", `${mod} was bound the wrong provision: ${bound.provision}`);
const pw = (await must(NODE, `cat ${secretPath}`)).trim();
assert.ok(bound.as && pw, `${mod}'s login or password was empty (as=${bound.as})`);
const conn = `postgresql://${bound.as}:${encodeURIComponent(pw)}@postgres:5432/${bound.as}?sslmode=disable`;
let pg = { out: "", ok: false };
const untilConn = Date.now() + 90_000;
while (Date.now() < untilConn) {
pg = await on(NODE, `docker exec mesh-postgres psql ${quote(conn)} -tAc 'select 1' 2>&1`);
if (pg.ok && /^1$/m.test(pg.out)) break;
if (/authentication failed/i.test(pg.out)) break;
await new Promise((r) => setTimeout(r, 3000));
}
assert.doesNotMatch(pg.out, /authentication failed/i, `postgres delivered ${mod} a password that does not authenticate:\n${pg.out}`);
assert.match(pg.out, /^1$/m, `${mod} could not connect to its granted postgres database as ${bound.as}:\n${pg.out}`);
}
// baserow also got its redis-cache binding: the mesh wrote the binding and unsealed the secret,
// and both arrived on the second node (a live redis AUTH is left to the redis single-module bed
// and the open provider-seal-key work).
const redisBound = await waitForBinding("/var/lib/baserow/redis.json");
assert.equal(redisBound.provision, "redis-cache", `baserow's redis binding is the wrong provision: ${redisBound.provision}`);
assert.ok(redisBound.as, `baserow's redis binding carries no login:\n${JSON.stringify(redisBound)}`);
const redisSecret = (await must(NODE, `cat /var/lib/baserow/redis.secret`)).trim();
assert.ok(redisSecret.length > 0, "baserow's redis secret was not delivered");
// Helper: wait for the mesh to write a consumer's bound file with an `as`, and parse it.
async function waitForBinding(path: string): Promise<{ as: string; provision: string }> {
let raw = "";
const until = Date.now() + 90_000;
while (Date.now() < until) {
const got = await on(NODE, `cat ${path} 2>/dev/null`);
if (got.ok && /"as"/.test(got.out)) { raw = got.out; break; }
await new Promise((r) => setTimeout(r, 3000));
}
assert.match(raw, /"as"/, `the mesh never wrote a binding with a login to ${path}:\n${raw}`);
return JSON.parse(raw) as { as: string; provision: string };
}
});