Rename mesh-control -> mesh-controller, substrate -> foundation

One name per thing, per the HQ glossary: the module/container/image/binary/repo
becomes mesh-controller, the seat the-controller, and the store+broker pair the
foundation (embedded base bundles, default template and example lock renamed with
their go:embed directives). No behaviour change — a pure vocabulary rename.

Claude-Session: https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx
This commit is contained in:
2026-09-16 18:40:40 +02:00
parent 49b80d8516
commit 5d6e8fbe7a
89 changed files with 785 additions and 785 deletions
+3 -3
View File
@@ -8,10 +8,10 @@
# The flow the test drives (OAuth stubbed, so it is the FLOW that is proven, not the vendor):
# the manager module seals the refresh token to the node's PUBLIC key -> the host unseals it and
# mounts the cleartext at the manager's bound path -> the manager calls the stub token endpoint ->
# submits back only { access token, re-sealed box } -> mesh-control seals the access token per
# submits back only { access token, re-sealed box } -> mesh-controller seals the access token per
# consumer holder -> the consumer runtime writes ~/.claude/.credentials.json, access-token-only.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# Build BOTH runtime images into the local daemon first (the scenario stocks and serves them by
# digest, which is where the host pulls them from):
# scripts/build-module-runtime.sh anthropic-manager /tmp/anthropic-manager.tar
@@ -35,7 +35,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# The two model-access runtimes, built by scripts/build-module-runtime.sh into the local daemon and
# loaded onto the machine, which holds them by their own image IDs.
- mesh-runtime-anthropic-manager:development
+2 -2
View File
@@ -1,6 +1,6 @@
# One machine that becomes a mesh and then assigns itself the audit logger.
#
# The substrate is first-node's — a store, a broker, the control plane — and one module image on
# The foundation is first-node's — a store, a broker, the control plane — and one module image on
# top: the tool runtime carrying the audit-logger (mesh-catalog). The node enrols itself and the
# mesh assigns it the audit logger, so its events account is one the mesh delivered, not the
# broker's own (novox/hq ADR 0048).
@@ -20,7 +20,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# The tool runtime with the audit-logger, built by scripts/build-runtime-image.sh into the local
# daemon and loaded onto the machine, which holds it by its own image ID.
- mesh-runtime-audit:development
+4 -4
View File
@@ -1,6 +1,6 @@
# One machine that becomes a mesh and is then assigned a SECOND wave of modules at once — the ones
# converted this session (novox/hq ADR 0039/0048/0052): mongodb, unifi and marrytts, with postgres
# carried along as a known-good control. This is catalogue-small's sibling: same first-node substrate,
# carried along as a known-good control. This is catalogue-small's sibling: same first-node foundation,
# same one-push co-residence, a different (and heavier) set of modules.
#
# The four exercise the three converted shapes:
@@ -14,7 +14,7 @@
# None of the four share a directory, so the shared-workspace refusal (novox/hq 04-ISSUES/012) does
# not bite; that class stays for a later bed.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# scripts/build-module-runtime.sh {mongodb,unifi,postgres} build the runtime images into the local
# daemon (mongodb carries mongosh; unifi and postgres carry their CLIs). The service images
# (mongo, unifi-controller, marytts, postgres) must be in the local daemon to be stocked.
@@ -31,13 +31,13 @@ machines:
egress: true
inbound: allow
# Sized up past catalogue-small's 6GiB: this wave carries two heavy JVM/embedded-DB service
# containers (the UniFi controller and MaryTTS) on top of the substrate, mongodb, postgres and
# containers (the UniFi controller and MaryTTS) on top of the foundation, mongodb, postgres and
# three node runtimes — a dozen containers, two of them memory-hungry at startup.
memory: 8GiB
cpus: 4
images:
- mesh-control:development
- mesh-controller:development
# The runtimes built by scripts/build-module-runtime.sh and stocked here. marrytts needs none.
- mesh-runtime-mongodb:development
- mesh-runtime-unifi:development
+3 -3
View File
@@ -16,7 +16,7 @@
# after enrol and before the push. Each module additionally OWNS its own config directory
# (/services/{sonarr,radarr}/config), which the mesh does create.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# scripts/build-module-runtime.sh {sonarr,radarr} build the two runtime images into the local daemon
# (they speak HTTP and need no CLI added). The service images lscr.io/linuxserver/{sonarr,radarr}
# must be in the local daemon to be stocked.
@@ -32,14 +32,14 @@ machines:
at: { segment: hosting, address: [192.0.2.10] }
egress: true
inbound: allow
# Two *arr apps (server + runtime each) on top of the first-node substrate — seven containers.
# Two *arr apps (server + runtime each) on top of the first-node foundation — seven containers.
# The Servarr images are lighter than catalogue-apps' JVM pair, so catalogue-small's 6GiB is
# ample headroom.
memory: 6GiB
cpus: 4
images:
- mesh-control:development
- mesh-controller:development
# The two runtimes built by scripts/build-module-runtime.sh and stocked here.
- mesh-runtime-sonarr:development
- mesh-runtime-radarr:development
+2 -2
View File
@@ -11,7 +11,7 @@
# the broker — so "the broker came up" is itself the proof the seed ran, because an unseeded store
# crash-loops the broker.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# scripts/build-module-runtime.sh mosquitto builds mesh-runtime-mosquitto:development (carrying
# mosquitto_ctrl and the compiled bootstrap entrypoint) into the local daemon, which this scenario
# pulls from the internet over its uplink. eclipse-mosquitto:2 must be in the local
@@ -32,7 +32,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
- mesh-runtime-mosquitto:development
place:
+4 -4
View File
@@ -3,14 +3,14 @@
#
# The per-backend beds each assign one module: postgres (a database provider), minio (an object-store
# provider), redis (a cache provider + tools), plex (a tools module). This raises the same first-node
# substrate and then assigns all four to the one anchor in a single push, so the proof is that they
# foundation and then assigns all four to the one anchor in a single push, so the proof is that they
# resolve and come up TOGETHER on one node — nothing new about any single module, everything new about
# their co-residence.
#
# The four are chosen because none of them share a directory, so the shared-workspace refusal
# (novox/hq 04-ISSUES/012 — the media stack) does not bite here; that class stays for a later bed.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# scripts/build-module-runtime.sh {postgres,redis,minio,plex} build the four runtime images into the
# local daemon (postgres carries psql, minio carries mc), which this scenario stocks and serves by
# the internet over its uplink. Only the mesh's own images come from the local daemon.
@@ -26,14 +26,14 @@ machines:
at: { segment: hosting, address: [192.0.2.10] }
egress: true
inbound: allow
# Sized up: this anchor runs the substrate (store, broker, control) plus four modules — three of
# Sized up: this anchor runs the foundation (store, broker, control) plus four modules — three of
# which are a server container and a runtime container each — so a dozen containers at once. The
# single-module beds run at 3GiB; co-residence needs the headroom.
memory: 6GiB
cpus: 4
images:
- mesh-control:development
- mesh-controller:development
# The four per-module runtimes, built by scripts/build-module-runtime.sh and stocked here. plex
# needs no server image in the lab — its runtime serves tools with no Plex to reach.
- mesh-runtime-postgres:development
+2 -2
View File
@@ -1,4 +1,4 @@
# One machine, raising a substrate from the bundle its host carries.
# One machine, raising a foundation from the bundle its host carries.
#
# This is the bootstrap class (novox/hq ADR 0009): no forge, no control plane to talk to, no
# delivery. It exists to develop the steps of raising a mesh on a machine that has nothing but a
@@ -24,7 +24,7 @@ machines:
# is what pinning asks for (novox/hq ADR 0006). The store and the broker are ordinary third-party
# images, and the machine pulls them from the internet like anything else.
images:
- mesh-control:development
- mesh-controller:development
place:
all: [host, runtime]
+4 -4
View File
@@ -11,7 +11,7 @@
# runtime and the host binary, which are prerequisites of the machine rather than parts of the
# mesh, and after that the mesh is on its own:
#
# - the substrate, the registry and the builder's own dependencies are PULLED from the internet,
# - the foundation, the registry and the builder's own dependencies are PULLED from the internet,
# which is where a bare machine gets them;
# - the control plane is BUILT, by the builder the installer carries, from a repository and a
# commit it is told to use;
@@ -26,7 +26,7 @@
#
# hosting (public, routable) home (private, behind the access point)
# anchor 192.0.2.20 ── anchor home-server 10.99.1.10 home server
# substrate, registry, workstation 10.99.1.20 workstation
# foundation, registry, workstation 10.99.1.20 workstation
# builder, control plane laptop 10.99.1.30 workstation
#
# EGRESS IS NOT OPTIONAL HERE. With nothing loaded, a sealed machine stops at the installer's first
@@ -35,7 +35,7 @@
# the public images, and the forge the control plane is cloned from.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BOOTSTRAP_BINARY=.../mesh-bootstrap
# MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# MESH_LAB_CATALOG=.../mesh-catalog/modules
# MESH_LAB_SOURCE=<forge url> MESH_LAB_SOURCE_REF=<commit>
scenario: fresh-mesh
@@ -74,7 +74,7 @@ machines:
cpus: 6
disk: 60GiB
# Three machines that JOIN. Host binary and a token, nothing else — no bootstrap, no substrate,
# Three machines that JOIN. Host binary and a token, nothing else — no bootstrap, no foundation,
# no registry. They are deliberately small: what they are here to prove is that a joined machine
# can be given a module the mesh built, which is a question about credentials and not about load.
home-server:
+5 -5
View File
@@ -2,10 +2,10 @@
#
# The smallest thing that proves a mesh can be raised. One machine on a public segment with a way
# out to the internet, the host placed, and nothing else — the installer carries the control
# plane's image and the substrate's own images are pulled over the uplink, exactly as they are on
# plane's image and the foundation's own images are pulled over the uplink, exactly as they are on
# a bare machine.
#
# The address matters: the substrate template names the broker at 192.0.2.10, and a token carries
# The address matters: the foundation template names the broker at 192.0.2.10, and a token carries
# that address verbatim as the endpoint an enrolling node dials. With one machine, that machine
# must BE it, or the mesh would hand out an endpoint nothing answers on.
scenario: genesis-single
@@ -23,7 +23,7 @@ machines:
# bare machine. A scenario that needs no images can omit this; genesis cannot.
egress: true
inbound: allow
# Enough for the substrate (store, broker), the registry, and two control planes during the
# Enough for the foundation (store, broker), the registry, and two control planes during the
# pivot. Smaller than the four-node bed's anchor, which also carries a whole service set.
memory: 8GiB
cpus: 4
@@ -33,8 +33,8 @@ machines:
#
# The runtime is not a mesh tier — it is a prerequisite of the machine, and the installer's first
# step refuses to go on without one. Placing it here is the lab preparing a machine, not the lab
# describing an installation. Everything above tier 0 — the substrate, the registry, the control
# describing an installation. Everything above tier 0 — the foundation, the registry, the control
# plane — is the installer's, and the lab places none of it. That is the whole point of this bed:
# if the lab placed the substrate, it would be describing installing all over again.
# if the lab placed the foundation, it would be describing installing all over again.
place:
all: [host, runtime]
+1 -1
View File
@@ -20,7 +20,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
- mesh-runtime-grafana:development
place:
+1 -1
View File
@@ -28,7 +28,7 @@ machines:
inbound: allow
images:
- mesh-control:development
- mesh-controller:development
place:
all: [host, runtime]
+8 -8
View File
@@ -1,28 +1,28 @@
# Two machines: one is the substrate, the other carries a lavinmq PROVIDER and a consumer of it.
# Two machines: one is the foundation, the other carries a lavinmq PROVIDER and a consumer of it.
#
# lavinmq is the mesh's own control-plane broker (mesh-broker), and it is ALSO offered as a
# user-facing capability: a module that needs a message queue gets its OWN broker — a scoped vhost and
# user on a lavinmq PROVIDER — not an account on the control broker. That makes lavinmq the sharp
# two-node case, for the same reason two-node-db is: the substrate's broker publishes 5672 on the node
# two-node case, for the same reason two-node-db is: the foundation's broker publishes 5672 on the node
# it runs on, and a lavinmq provider must publish 5672 too for its consumers to reach it — so the two
# cannot share a machine. The moment a real amqp provider must own a node's 5672, it collides with the
# control broker already there, and the chain is blocked single-node.
#
# This is the split that unblocks it. `anchor` runs the substrate (store, broker, control) and NOTHING
# This is the split that unblocks it. `anchor` runs the foundation (store, broker, control) and NOTHING
# else. `laptop` runs the whole chain: the lavinmq PROVIDER and the amqp-ping CONSUMER that requires
# it. Provider and consumer are co-located on laptop, so the grant never crosses a node boundary; only
# enrolment crosses to anchor, over the underlay both machines share. And because the substrate broker
# enrolment crosses to anchor, over the underlay both machines share. And because the foundation broker
# is on the OTHER node, the provider owns laptop's 5672 uncontested.
#
# It needs a host binary and the substrate bundle:
# It needs a host binary and the foundation bundle:
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
#
# HELPER — stock the two runtimes into the local daemon before the run (some may already be there):
# scripts/build-module-runtime.sh lavinmq /tmp/lavinmq.tar
# scripts/build-module-runtime.sh amqp-ping /tmp/amqp-ping.tar
# The service image cloudamqp/lavinmq:latest must be in the local daemon too — it is already stocked as
# the substrate's own broker image; each node pulls what it runs from the internet, over its own
# the foundation's own broker image; each node pulls what it runs from the internet, over its own
# uplink.
scenario: lavinmq-bed
@@ -46,7 +46,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# The lavinmq provider's runtime (reused for its run-once bootstrap and its provisioner) and the
# amqp-ping consumer's runtime, both built by scripts/build-module-runtime.sh into the local daemon
# and loaded onto the machines, which hold them by their own image IDs.
+3 -3
View File
@@ -11,9 +11,9 @@
# openai.env carries OPENAI_BASE_URL=http://<at>:<port>/v1. The test asserts that URL and that a
# request to it reaches the running model server.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# Both modules are pure declaration (a server container + a templated file) — no module runtime image
# is built; the bed only stocks the substrate and the ollama server image.
# is built; the bed only stocks the foundation and the ollama server image.
scenario: local-model-bed
segments:
@@ -31,7 +31,7 @@ machines:
disk: 40GiB
images:
- mesh-control:development
- mesh-controller:development
place:
all: [host, runtime]
+1 -1
View File
@@ -21,7 +21,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# minio's runtime, built by scripts/build-module-runtime.sh minio (it carries mc), loaded onto
# the machine.
- mesh-runtime-minio:development
+8 -8
View File
@@ -1,7 +1,7 @@
# The usage context store, proved end to end (novox/hq ADR 0054). A postgres PROVIDER and the
# model-usage CONSUMER ride one node; the substrate (store, broker, control) owns the other. As in
# two-node-db, the app-postgres provider and the mesh's own substrate store both want host port 5432,
# so they cannot share a machine — the substrate lives on `anchor` and NOTHING else, and `laptop`
# model-usage CONSUMER ride one node; the foundation (store, broker, control) owns the other. As in
# two-node-db, the app-postgres provider and the mesh's own foundation store both want host port 5432,
# so they cannot share a machine — the foundation lives on `anchor` and NOTHING else, and `laptop`
# runs postgres plus model-usage. Provider and consumer are co-located on laptop, so only enrolment
# crosses to anchor, over the underlay both machines already share.
#
@@ -9,7 +9,7 @@
# provisioned postgres store, at BOTH grains (licence and session, differing only in `consumer`),
# LATEST-per-key, and IN THE CLEAR — an ordinary select returns the numeric value and its raw payload.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# Build BOTH runtime images into the local daemon first (the scenario stocks and serves them by
# digest, which is where the host pulls them from):
# scripts/build-module-runtime.sh postgres /tmp/postgres.tar
@@ -22,7 +22,7 @@ segments:
cidr: [192.0.2.0/24]
machines:
# The substrate ONLY: store, broker, control — three containers.
# The foundation ONLY: store, broker, control — three containers.
anchor:
at: { segment: hosting, address: [192.0.2.10] }
egress: true
@@ -30,8 +30,8 @@ machines:
memory: 4GiB
cpus: 4
# The postgres PROVIDER (server + broker-bound runtime) and the model-usage CONSUMER (its run-once
# migrate and its long-lived event runtime). The 5432-vs-substrate conflict is gone because the
# substrate store is on the OTHER node. The runtime images plus postgres:17-alpine are pulled from
# migrate and its long-lived event runtime). The 5432-vs-foundation conflict is gone because the
# foundation store is on the OTHER node. The runtime images plus postgres:17-alpine are pulled from
# the internet over the uplink; forty gigabytes holds them with room to spare.
laptop:
at: { segment: hosting, address: [192.0.2.20] }
@@ -42,7 +42,7 @@ machines:
disk: 40GiB
images:
- mesh-control:development
- mesh-controller:development
# The per-module runtimes, built by scripts/build-module-runtime.sh and stocked here. Each carries
# its module's code — postgres its provisioner, model-usage its consumer, tools and run-once migrate.
- mesh-runtime-postgres:development
+5 -5
View File
@@ -2,7 +2,7 @@
#
# The common case, and the one worth getting right first: a person with a single machine runs the
# installer and ends up with a mesh that works. Not a mesh that *runs* — `17-raising-a-mesh` is
# careful about that difference, and so is this scenario. Genesis ends with a substrate, a registry,
# careful about that difference, and so is this scenario. Genesis ends with a foundation, a registry,
# a built control plane and a builder, and a mesh in that state cannot produce anything and holds no
# record of what it has. Calling that "up" is how the catalogue came to be missing from a test for
# weeks without anything complaining.
@@ -10,11 +10,11 @@
# So the test driving this asks the harder question: can this machine, given nothing but a container
# runtime and the host binary, end up holding
#
# - a substrate and a registry it pulled from the internet,
# - a foundation and a registry it pulled from the internet,
# - a control plane it BUILT, and then rebuilt from its own repository through the module path,
# - a builder that takes work over the broker,
# - the shared base every module with code of its own stands on,
# - a store of its own — the substrate's is the control plane's own plumbing, not a provider,
# - a store of its own — the foundation's is the control plane's own plumbing, not a provider,
# - a catalogue, so it can say what it has and what a change reaches,
# - and a module of its own, built, provisioned and running.
#
@@ -26,7 +26,7 @@
# EGRESS IS NOT OPTIONAL. With nothing loaded, a sealed machine stops at the installer's first pull.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BOOTSTRAP_BINARY=.../mesh-bootstrap
# MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# MESH_LAB_CATALOG=.../mesh-catalog/modules
# MESH_LAB_SOURCE=<forge url> MESH_LAB_SOURCE_REF=<commit>
scenario: one-node-mesh
@@ -37,7 +37,7 @@ segments:
cidr: [192.0.2.0/24]
machines:
# The address matters: the substrate template names the broker at a fixed address, and a token
# The address matters: the foundation template names the broker at a fixed address, and a token
# carries that verbatim as the endpoint an enrolling node dials. With one machine, that machine
# must BE it, or the mesh hands out an endpoint nothing answers on.
#
+2 -2
View File
@@ -11,7 +11,7 @@
# OPENAI_API_KEY (env file + the Codex auth.json). The test asserts the written key equals the one
# the operator set — through the sealed delivery path, unchanged.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# Build the consumer runtime image into the local daemon first (raise() stocks it from there):
# scripts/build-module-runtime.sh openai-consumer /tmp/openai-consumer.tar
scenario: openai-bed
@@ -30,7 +30,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# The static-key consumer runtime, built by scripts/build-module-runtime.sh into the local daemon and
# loaded onto the machine, which holds it by its own image ID.
- mesh-runtime-openai-consumer:development
+2 -2
View File
@@ -1,7 +1,7 @@
# One machine that becomes a mesh and then assigns itself plex's tool runtime.
#
# The audit-node bed proved an assigned *consumer* (novox/hq ADR 0048). This proves an assigned
# module that *serves tools* (ADR 0052): the same first-node substrate, plus plex's tool runtime on
# module that *serves tools* (ADR 0052): the same first-node foundation, plus plex's tool runtime on
# top. The node enrols itself, the mesh issues plex a broker account scoped to serve.plex.* and
# assigns it, the host runs the runtime container, and a caller invokes plex.plex_reachable over the
# mesh — proof the module runs its own code as its own process under its own scoped account.
@@ -21,7 +21,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# Plex's tool runtime, built by scripts/build-module-runtime.sh plex into the local daemon and
# loaded onto the machine, which holds it by its own image ID.
- mesh-runtime-plex:development
+1 -1
View File
@@ -20,7 +20,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# postgres's runtime, built by scripts/build-module-runtime.sh postgres (it carries psql), loaded
# onto the machine.
- mesh-runtime-postgres:development
+1 -1
View File
@@ -20,7 +20,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# Redis's tool+provisioner runtime, built by scripts/build-module-runtime.sh redis into the local
# daemon and loaded onto the machine, which holds it by its own image ID.
- mesh-runtime-redis:development
+4 -4
View File
@@ -14,9 +14,9 @@
# publicly-trusted certificate and answering an HTTP-01 challenge at the name — is proven separately
# by certificates.test.ts against a real ACME server (Pebble), driving the same proxy binary.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# scripts/build-route-proxy-image.sh builds mesh-route-proxy:development into the local daemon
# (from mesh-control/examples/route-proxy, via mesh-catalog/modules/route-proxy/Dockerfile).
# (from mesh-controller/examples/route-proxy, via mesh-catalog/modules/route-proxy/Dockerfile).
# alpine:latest must be in the local daemon — hello-web's backend is a bare alpine that serves a
# fixed page over a busybox nc loop. Both images are stocked and served by digest.
scenario: route-forwarding
@@ -35,8 +35,8 @@ machines:
cpus: 2
images:
- mesh-control:development
# The route-proxy's image, built from the canonical Go proxy in mesh-control by
- mesh-controller:development
# The route-proxy's image, built from the canonical Go proxy in mesh-controller by
# scripts/build-route-proxy-image.sh, and hello-web's backend, a bare alpine nc loop.
- mesh-route-proxy:development
+2 -2
View File
@@ -14,7 +14,7 @@
# - the container fires when the cron is due (top of the next minute);
# - it fires AGAIN on the following minute — recurrence, not a one-shot.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_MODULES=.../mesh-control/examples/modules
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_MODULES=.../mesh-controller/examples/modules
# alpine:latest must be in the local daemon; the machine pulls it from the internet over its
# uplink, and the scheduled container declares it exactly as the catalogue writes it.
# There is no runtime image: schedtest carries no code of its own — the scheduled container is a
@@ -35,7 +35,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
place:
# Only the host — schedtest has no mesh-runtime to place. The tick image is pulled from the
+1 -1
View File
@@ -20,7 +20,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
- mesh-runtime-sonarr:development
place:
+2 -2
View File
@@ -9,7 +9,7 @@
# built lazily and never throws at registration, so the runtime logs `[mesh-tools] serving 3 tool(s)`
# and binds its serve queues regardless; a tool would only fail if it were actually invoked.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# scripts/build-module-runtime.sh confluence builds mesh-runtime-confluence:development into the
# local daemon, which the machine pulls from the internet over its uplink. confluence
# needs no service image — it is tools-only.
@@ -29,7 +29,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# confluence's runtime, built by scripts/build-module-runtime.sh confluence into the local daemon
# and loaded onto the machine, which holds it by its own image ID. There is no
# service image: confluence is tools-only and outbound-only.
+2 -2
View File
@@ -10,7 +10,7 @@
# throws at registration, so the runtime logs `[mesh-tools] serving 23 tool(s)` and binds its serve
# queues regardless; a tool would only fail if it were actually invoked without real creds.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# scripts/build-module-runtime.sh gitlab builds mesh-runtime-gitlab:development into the local
# daemon, which the machine pulls from the internet over its uplink. gitlab needs no
# service image — it is tools-only.
@@ -30,7 +30,7 @@ machines:
cpus: 2
images:
- mesh-control:development
- mesh-controller:development
# gitlab's runtime, built by scripts/build-module-runtime.sh gitlab into the local daemon and
# loaded onto the machine, which holds it by its own image ID. There is no
# service image: gitlab is tools-only and outbound-only.
+6 -6
View File
@@ -1,13 +1,13 @@
# The DB-consumer chain a single node cannot host, proved across two machines.
#
# The app-postgres provider and the mesh's own substrate store both want host port 5432, so they
# cannot share a machine — the collision that blocked this chain single-node. Here the substrate
# The app-postgres provider and the mesh's own foundation store both want host port 5432, so they
# cannot share a machine — the collision that blocked this chain single-node. Here the foundation
# (store, broker, control) lives on `anchor` and NOTHING else; `laptop` runs the whole chain —
# postgres and redis PROVIDERS plus the baserow and letta CONSUMERS that require them. Both
# machines sit on one shared segment and enrol into the one mesh; only enrolment crosses to anchor,
# over the underlay both machines already share. Provider and consumers are co-located on laptop, so
# no cross-node module comms and no overlay are needed — and the 5432-vs-substrate conflict is gone
# because the substrate store is on the OTHER node.
# no cross-node module comms and no overlay are needed — and the 5432-vs-foundation conflict is gone
# because the foundation store is on the OTHER node.
scenario: two-node-db
segments:
@@ -16,7 +16,7 @@ segments:
cidr: [192.0.2.0/24]
machines:
# The substrate ONLY: store, broker, control — three containers. Four gigabytes is plenty for a
# The foundation ONLY: store, broker, control — three containers. Four gigabytes is plenty for a
# node that hosts no modules; the thrash the two-nodes bed warns of comes from stacking eleven
# containers on a node, which this one never does.
anchor:
@@ -43,7 +43,7 @@ machines:
disk: 60GiB
images:
- mesh-control:development
- mesh-controller:development
# The per-module runtimes, built by scripts/build-module-runtime.sh and stocked here. Each carries
# its module's provisioner, so no separate mesh-provision-* image is listed — the runtime is the
# provisioner (ADR 0048).
+2 -2
View File
@@ -17,7 +17,7 @@ machines:
at: { segment: hosting, address: [192.0.2.10] }
egress: true
inbound: allow
# The whole substrate, the registry, the builder, an adopted workload and the modules under
# The whole foundation, the registry, the builder, an adopted workload and the modules under
# test all land here — eleven containers before the forge arrives. At the 1GiB default this
# machine thrashes, and it presents as "the mesh hangs": every exec slows from 15s to 105s
# and the forge test fails on a status poll that is merely queued behind page-outs.
@@ -30,7 +30,7 @@ machines:
memory: 2GiB
images:
- mesh-control:development
- mesh-controller:development
# And the builder, because it is a module the mesh assigns rather than a program somebody
# starts by hand — which is the only way its credential can be one the mesh delivered.
- mesh-builder:development
+7 -7
View File
@@ -1,15 +1,15 @@
# The whole `ace` server's converted service set, installed together on ONE node behind the mesh
# substrate — the media/home-automation half of the whole-mesh rehearsal (novox/hq). Sibling of
# foundation — the media/home-automation half of the whole-mesh rehearsal (novox/hq). Sibling of
# scenarios/whole-mesh-novox.yml; same topology, a different (larger, media-heavy) module set.
#
# Substrate (store, broker, control) rides `anchor` and NOTHING else; ALL of ace's services ride the
# Foundation (store, broker, control) rides `anchor` and NOTHING else; ALL of ace's services ride the
# `ace` node. An overlay is placed so the two DB consumers (baserow, letta) reach the postgres/redis
# providers co-located with them. The media stack (sonarr/radarr/lidarr/plex/bazarr/nzbget/
# qbittorrent/bookshelf) shares the operator-owned library directories under /services/media (ADR
# 0051 `accesses`); the test pre-creates them on the node, as the operator would, before the push —
# the mesh confirms the paths exist and mounts them, but creates and chowns none of it.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# The runtimes are built by scripts/build-module-runtime.sh (one per module) and must be in the local
# daemon, because nothing serves them and nothing can. Every media/app image is pulled from the
# internet by the node itself, over its uplink, by the digest its module.json already pins. The test
@@ -30,9 +30,9 @@ machines:
memory: 4GiB
cpus: 4
disk: 20GiB
# The substrate only, so the control plane's image only. The runtimes belong on the node that
# The foundation only, so the control plane's image only. The runtimes belong on the node that
# runs the modules, and a 20GiB disk has no room for them anyway.
images: [mesh-control:development]
images: [mesh-controller:development]
# The whole ace service set — 24 modules, ~50 containers, several heavy (Plex, Home Assistant,
# Letta ~1.8GiB, Baserow ~1.5GiB, the UniFi controller's JVM, mssql ~2GiB). Sized past novox.
ace:
@@ -42,7 +42,7 @@ machines:
memory: 18GiB
cpus: 8
disk: 120GiB
# Every runtime. Not mesh-control: the control plane runs on the anchor.
# Every runtime. Not mesh-controller: the control plane runs on the anchor.
images:
- mesh-runtime-postgres:development
- mesh-runtime-redis:development
@@ -70,7 +70,7 @@ machines:
- mesh-runtime-unifi:development
images:
- mesh-control:development
- mesh-controller:development
# The per-module runtimes (built by scripts/build-module-runtime.sh).
- mesh-runtime-postgres:development
- mesh-runtime-redis:development
+12 -12
View File
@@ -1,19 +1,19 @@
# The FULL mesh in its REAL production shape: two segments, one access point, one overlay.
#
# This is the first multi-segment whole-mesh bed. The earlier flat whole-mesh-full sat every node
# on one public segment with a SEPARATE `anchor` carrying the substrate. Production is not flat, and
# on one public segment with a SEPARATE `anchor` carrying the foundation. Production is not flat, and
# there is no separate anchor: `novox` IS the anchor. It sits on the routable `hosting` segment,
# runs the substrate (store, broker, control) AND its own service set AND is the overlay hub and the
# runs the foundation (store, broker, control) AND its own service set AND is the overlay hub and the
# public ingress. `ace`, `shanks` and `g14` sit on the household `home` segment BEHIND a NAT gateway
# — the access point — reachable from the outside only through what they dial out to.
#
# hosting (public, routable) home (private, behind the access point)
# novox 192.0.2.20 ── anchor ace 10.99.1.10 home server, media/IoT set
# substrate + novox set shanks 10.99.1.20 workstation (light)
# foundation + novox set shanks 10.99.1.20 workstation (light)
# overlay hub, ingress g14 10.99.1.30 workstation (light)
#
# The `home` gateway masquerades v4 outbound and forwards inbound (an ordinary household router).
# Home nodes reach novox's public 192.0.2.20 by dialling OUT through it: the substrate broker (5671),
# Home nodes reach novox's public 192.0.2.20 by dialling OUT through it: the foundation broker (5671),
# the mesh's own artifact store, and — the thing this bed exists to prove — the WireGuard overlay hub
# (51820/udp). The hub keepalive holds the NAT hole open so the tunnel, once formed, stays up. novox
# cannot initiate to a home node at all; every home↔novox path is either the overlay or a forwarded
@@ -35,20 +35,20 @@
# the gateway's masquerade? The driving test verifies the WireGuard handshake and cross-segment
# reachability over the overlay explicitly, and reports form-vs-break as its headline.
#
# Substrate-on-novox collides on two host ports the separate-anchor beds never hit: the substrate
# store binds 127.0.0.1:5432 and novox's postgres provider publishes 5432; the substrate broker binds
# Foundation-on-novox collides on two host ports the separate-anchor beds never hit: the foundation
# store binds 127.0.0.1:5432 and novox's postgres provider publishes 5432; the foundation broker binds
# 5671 + 127.0.0.1:5672 and novox's lavinmq provider publishes 5672. The driving test REMAPS those two
# provider host publishes off the substrate's ports (consumers reach the providers over the mesh
# provider host publishes off the foundation's ports (consumers reach the providers over the mesh
# network on the container port, so the host side is free to move). Reported as a topology finding.
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# MESH_LAB_BOOTSTRAP_BINARY=.../mesh-bootstrap MESH_LAB_CATALOG=.../mesh-catalog/modules
#
# GENESIS AND JOINING ARE TWO DIFFERENT ACTS, and this bed distinguishes them. novox is brought
# into existence by `mesh-bootstrap` — the same program a bare machine runs — and is afterwards a
# working mesh of one, with a registry and a control plane that is an ordinary module pinned to an
# image that registry serves. ace, shanks and g14 then JOIN it: host binary, token, enrol, run. No
# bootstrap, no substrate, no registry. novox is never enrolled twice, because the installer
# bootstrap, no foundation, no registry. novox is never enrolled twice, because the installer
# already did it.
#
# The images: are the UNION of the novox set (feat/novox-conversions @ 431310f: the slug + roundcube
@@ -76,8 +76,8 @@ segments:
mapping_ttl: 120s
machines:
# The anchor: substrate (store, broker, control) + the whole novox service set + overlay hub +
# public ingress. Bigger than the flat bed's novox, because it now carries the substrate too.
# The anchor: foundation (store, broker, control) + the whole novox service set + overlay hub +
# public ingress. Bigger than the flat bed's novox, because it now carries the foundation too.
novox:
at: { segment: hosting, address: [192.0.2.20] }
egress: true
@@ -90,7 +90,7 @@ machines:
# because they carry no runtime image of their own — what they run is third-party or is the
# node itself.
#
# **mesh-control is NOT here, and its absence is the point** (novox/hq ADR 0067). The anchor is
# **mesh-controller is NOT here, and its absence is the point** (novox/hq ADR 0067). The anchor is
# brought into existence by the installer, and the installer carries the control plane's image
# inside itself — that is the whole reason a machine that can reach no registry can still raise
# a mesh. Handing it over from the workstation as well would mean the bed never found out
+9 -9
View File
@@ -1,10 +1,10 @@
# The whole `novox` server's converted service set, installed together on ONE node behind the mesh
# substrate — the whole-catalogue install the rebuild has never actually run. First stage of a
# foundation — the whole-catalogue install the rebuild has never actually run. First stage of a
# whole-mesh rehearsal (novox/hq).
#
# Topology, proven by test/integration/assigned-two-node-db.test.ts: the substrate (store, broker,
# Topology, proven by test/integration/assigned-two-node-db.test.ts: the foundation (store, broker,
# control) rides `anchor` and NOTHING else; ALL of novox's services ride the `novox` node — its own
# postgres provider owns 5432 there, so it cannot co-locate with the substrate store on 5432. Both
# postgres provider owns 5432 there, so it cannot co-locate with the foundation store on 5432. Both
# machines sit on one public segment and enrol into the one mesh; an overlay is placed so a
# consumer's binding `at` resolves to novox's private address and every consumer reaches the
# providers co-located with it.
@@ -15,7 +15,7 @@
# apps portainer verdaccio registry route-proxy mailu
# node-level firewall fail2ban
#
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/substrate-first-node.lock
# MESH_LAB_HOST_BINARY=.../mesh-host MESH_LAB_BUNDLE=.../examples/foundation-first-node.lock
# The runtimes are built by scripts/build-module-runtime.sh (one per module that has code) and the
# route-proxy image by scripts/build-route-proxy-image.sh; those must be in the local daemon,
# because nothing serves them and nothing can. Every third-party image is pulled from the internet
@@ -29,7 +29,7 @@ segments:
cidr: [192.0.2.0/24]
machines:
# The substrate ONLY: store, broker, control. Nothing else lands here.
# The foundation ONLY: store, broker, control. Nothing else lands here.
anchor:
at: { segment: hosting, address: [192.0.2.10] }
egress: true
@@ -37,9 +37,9 @@ machines:
memory: 4GiB
cpus: 4
disk: 20GiB
# The substrate only, so the control plane's image only. Handing this machine the whole set of
# The foundation only, so the control plane's image only. Handing this machine the whole set of
# runtimes would fill a 20GiB disk with images nothing on it will ever start.
images: [mesh-control:development]
images: [mesh-controller:development]
# The whole novox service set — ~38 containers (five providers with runtimes, six consumers with
# runtimes, portainer/verdaccio/registry/route-proxy, the nine-container Mailu stack and its
# runtime) plus two node-level modules. mssql alone wants ~2GiB; Mailu, Nextcloud and Keycloak are
@@ -55,7 +55,7 @@ machines:
# layers and the runtimes. A hundred gigabytes holds the whole set without exhausting the disk
# mid-apply.
disk: 100GiB
# Every runtime, and the proxy. Not mesh-control: the control plane runs on the anchor.
# Every runtime, and the proxy. Not mesh-controller: the control plane runs on the anchor.
images:
- mesh-runtime-postgres:development
- mesh-runtime-redis:development
@@ -73,7 +73,7 @@ machines:
- mesh-route-proxy:development
images:
- mesh-control:development
- mesh-controller:development
# The per-module runtimes (built by scripts/build-module-runtime.sh). registry, route-proxy,
# invoicing, firewall and fail2ban carry no mesh-runtime image; route-proxy ships its own.
- mesh-runtime-postgres:development