From 751948f0f940634effd86a8e579bf794dff85f2c Mon Sep 17 00:00:00 2001 From: jochen Date: Thu, 10 Sep 2026 23:16:05 +0200 Subject: [PATCH] The lab had a registry that production does not, so it tested a fiction MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The lab raised a `registry` VM, pushed ~73 images into it from the workstation, and rewrote every manifest reference — third-party ones included — to point at it. No production mesh has such a thing. So every bed proved that a machine could fetch an image from a registry that exists nowhere else, and the bootstrap problems that only appear when a machine has to fetch for itself went unfound. What replaces it is the two things that are true in the world: **Public images come from the public internet.** mesh-lab already created a NAT'd uplink for exactly this and attached it to any machine declaring `egress`; no scenario ever declared it. They do now, and third-party references are left exactly as the catalogue writes them. **The mesh's own images have no registry and never will.** mesh-control, mesh-builder, mesh-route-proxy and the per-module runtimes are built from source and exist in no registry. A machine gets them the way an operator's machine does — they are built here and loaded onto it — and is then named by the digest of its own image configuration, which mesh-host now accepts as "an image this machine already holds". `images:` therefore means only *ours*, and a third-party entry is refused rather than quietly loaded: otherwise the fiction returns one convenient line at a time. It is per-machine as well, because "everything, everywhere" was never a description of anything real — handing whole-mesh-full's union to its two 30GiB workstations would fill the disk with runtimes nothing on them will start. **The uplink and the declared gateway would have fought, silently.** A gateway container and the transit router reach the scenario and nothing else; a default route through either is a black hole for anything outside, and it beats the uplink's DHCP route on metric. So a machine with egress states the scenario's ranges explicitly — through the same gateway or transit it would have defaulted to, so the overlay-across-NAT path is unchanged — and leaves the default to the uplink. A range with no path inside the scenario becomes `unreachable` rather than falling through: 192.168.1.0/24 is an ordinary private range in fact, and letting it escape would put scenario traffic on whatever network the workstation is sitting on. `scenarioRoutesFor` is pure and tested, because a decision only a full raise could check is one nobody checks. The registry-reachability check the raise gained earlier is kept, pointed at the real thing: every machine with egress must resolve a name and reach the internet before the raise says it finished. Same failure it was written for — a raise that returns, an apply that dies on its first pull, an instance left a bare shell — now guarding the path that actually carries. The base image's trust of the documentation ranges as plain-HTTP registries STAYS. It was never only for the lab's registry: the mesh has one of its own, the `registry` module, serving artifacts to the whole mesh over plain HTTP from whatever node runs it. Claude-Session: https://claude.ai/code/session_01LrgweAeERJYBg88c5cKDzF --- src/cli.ts | 10 +- src/declaration/parse.ts | 3 + src/declaration/types.ts | 29 ++- src/declaration/validate.ts | 38 +-- src/lifecycle/address.ts | 119 +++++++++ src/lifecycle/base.ts | 21 +- src/lifecycle/egress.ts | 126 ++++++++++ src/lifecycle/place.ts | 134 +++++++++- src/lifecycle/raise.ts | 64 ++--- src/lifecycle/registry.ts | 487 ------------------------------------ src/pinning.ts | 102 +++++--- src/suite.ts | 6 +- src/warm.ts | 21 +- 13 files changed, 558 insertions(+), 602 deletions(-) create mode 100644 src/lifecycle/egress.ts delete mode 100644 src/lifecycle/registry.ts diff --git a/src/cli.ts b/src/cli.ts index e353173..5b624ee 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -233,10 +233,12 @@ async function main(): Promise { const seconds = ((Date.now() - started) / 1000).toFixed(1); console.log(`\nraised ${raised.instanceId} in ${seconds}s — ${raised.machines.length} machines usable`); if (raised.images.length > 0) { - // Printed because this is what a declaration pins, and it is not knowable until the - // scenario has been raised — the digest belongs to this registry. - console.log(`\nimages served, pinned by digest:`); - for (const image of raised.images) console.log(` ${image}`); + // Printed because this is what a declaration names them by, and it is not knowable until + // the image has been built — an image ID is the digest of its own configuration. + console.log(`\nthe mesh's own images, as the machines now hold them:`); + for (const image of raised.images) { + console.log(` ${image.requested} → ${image.reference}`); + } } if (scenario.snapshot) { const took = await snapshot(raised.instanceId, scenario.snapshot); diff --git a/src/declaration/parse.ts b/src/declaration/parse.ts index 38c490b..d1bd5a2 100644 --- a/src/declaration/parse.ts +++ b/src/declaration/parse.ts @@ -41,6 +41,9 @@ function normaliseMachine(raw: unknown): Machine { if (machine["memory"] !== undefined) result.memory = String(machine["memory"]); if (machine["cpus"] !== undefined) result.cpus = Number(machine["cpus"]); if (machine["disk"] !== undefined) result.disk = String(machine["disk"]); + // Absent and empty are different: absent means "all of the scenario's images", an explicit empty + // list means "none". A machine that runs nothing of ours should be able to say so. + if (machine["images"] !== undefined) result.images = toList(machine["images"]); return result; } diff --git a/src/declaration/types.ts b/src/declaration/types.ts index 343ef28..ae71d9f 100644 --- a/src/declaration/types.ts +++ b/src/declaration/types.ts @@ -110,6 +110,19 @@ export interface Machine { */ memory?: string; cpus?: number; + /** + * Which of the scenario's `images:` this machine is handed. + * + * Absent means all of them, which is right for a one-machine bed and wrong for a mesh: a + * workstation running one small module does not want forty runtimes copied onto a 30GiB disk. + * That is not a lab economy, it is what is true — an operator's machine holds the images its own + * modules need, because somebody put them there. + * + * Every entry must appear in the scenario's `images:`. Naming one that does not is refused + * rather than ignored, because a machine silently missing an image fails much later, inside an + * apply, as a container that will not start. + */ + images?: string[]; /** * Root disk size, e.g. "60GiB". Left unset, the VM uses the storage pool's default, which is * enough for a handful of modules. A broad install that stocks many runtime + service images @@ -142,11 +155,19 @@ export interface Scenario { policy?: Policy[]; place?: Placement; /** - * Container images this scenario needs inside it. + * **The mesh's own images** — the ones that exist in no registry and are put onto a machine by + * whoever built them. * - * A sealed machine cannot reach a registry, so the lab raises one on a public segment and - * serves these from it. Written as tags — the digest a declaration pins is the one THIS - * registry assigns, and it is reported when the scenario is raised. + * mesh-control, mesh-builder, mesh-route-proxy, the per-module runtimes and the provisioners are + * built from source and published nowhere. A machine gets them the way an operator's machine + * does: they are built on the workstation, loaded onto the machine, and named by the digest of + * their own image configuration. Written as tags, because a tag is what `docker save` can + * export; what a declaration then pins is the image ID, reported when the scenario is raised. + * + * **Third-party images do not belong here.** postgres, gitea, the mailu stack and everything + * else are pulled from the internet over a machine's `egress` uplink, exactly as they are in + * production. The lab used to serve them from a registry of its own, and that registry did not + * exist anywhere else — so every bootstrap problem it papered over went unfound. */ images?: string[]; /** Name the state once placement finishes, so a run can return to it. */ diff --git a/src/declaration/validate.ts b/src/declaration/validate.ts index ee6f2da..bccf643 100644 --- a/src/declaration/validate.ts +++ b/src/declaration/validate.ts @@ -15,6 +15,7 @@ import type { Scenario, Segment } from "./types.ts"; import { contains, familyOf, parseAddress, parseCidr, type Cidr } from "./net.ts"; +import { isMeshBuilt } from "../pinning.ts"; /** RFC 5737 and RFC 3849. The only addresses guaranteed never to route on the real internet. */ const DOCUMENTATION_RANGES = [ @@ -315,27 +316,36 @@ export function validate(scenario: Scenario): void { } } - for (const image of scenario.images ?? []) { + const declaredImages = scenario.images ?? []; + for (const image of declaredImages) { if (!image.trim()) { problems.push("images: an empty entry names nothing"); } else if (image.includes("@sha256:")) { - // The digest a declaration pins is the one the LAB's registry assigns, which is not - // knowable before the scenario is raised. Naming an upstream digest here would pin - // something this registry will never serve. + // A tag, because a tag is what `docker save` exports. The reference a declaration ends up + // using is the image's own ID, which is not knowable until the image has been built. problems.push( - `images: '${image}' is pinned by digest. Name it by tag — the lab's registry assigns ` + - `its own digest and reports it when the scenario is raised`, + `images: '${image}' is pinned by digest. Name it by tag — what a declaration uses is the ` + + `ID of the image loaded onto the machine, reported when the scenario is raised`, + ); + } else if (!isMeshBuilt(image)) { + // **The rule that replaced the lab's registry.** Anything with somewhere to be fetched from + // is fetched from there, over the machine's uplink, exactly as in production. Serving it + // from inside the scenario instead is what hid the bootstrap faults this lab exists to find. + problems.push( + `images: '${image}' is not one of the mesh's own images, so nothing loads it. It is ` + + `pulled from the internet by the machine that needs it — give that machine 'egress: true' ` + + `and delete this line. Only mesh-* images, which exist in no registry, are placed by hand`, ); } } - if ((scenario.images ?? []).length > 0) { - const hasPublicV4 = Object.values(scenario.segments) - .some((s) => s.kind === "public" && s.cidr.some((c) => !c.includes(":"))); - if (!hasPublicV4) { - problems.push( - "images: this scenario declares images and has no public IPv4 segment to serve them " + - "from. The registry stands in for the outside world, so it sits on a public segment", - ); + for (const [name, machine] of Object.entries(scenario.machines)) { + for (const image of machine.images ?? []) { + if (!declaredImages.includes(image)) { + problems.push( + `machines.${name}.images: '${image}' is not in this scenario's images:. A machine can ` + + `only be handed one of the images the scenario says it has`, + ); + } } } diff --git a/src/lifecycle/address.ts b/src/lifecycle/address.ts index 6d6ed48..3bb9ed0 100644 --- a/src/lifecycle/address.ts +++ b/src/lifecycle/address.ts @@ -180,6 +180,20 @@ function withPrefix(scenario: Scenario, segment: string, address: string): strin * * The router's inside address is the first host address of the range, chosen rather than * declared because a scenario has nothing to say about it. + * + * **A machine with `egress` is routed differently, and it has to be.** The scenery inside a + * scenario — a gateway container, the transit router — reaches the scenario and nothing else: it + * has no route to the real internet, and never will, because it exists to reproduce a household + * router rather than to be one. So a default route pointing at it is a black hole for anything + * outside, and it wins over the uplink's DHCP route on metric. A machine that must pull an image + * would then sit there failing, with a default route that looks perfectly reasonable. + * + * So an egress machine keeps the uplink as its default and gets an EXPLICIT route to every other + * segment in the scenario, through the same gateway or transit it would otherwise have defaulted + * to. Where there is no such path, the range is made `unreachable` rather than left to fall + * through: 192.168.1.0/24 in a scenario is a documentation range in spirit but an ordinary private + * one in fact, and letting it escape to the uplink would put scenario traffic on whatever network + * the workstation happens to sit on. */ export async function applyDefaultRoutes( scenario: Scenario, @@ -195,6 +209,13 @@ export async function applyDefaultRoutes( // segment routes through transit instead — otherwise it can reach its own network and // nothing else, which is not what being on the internet means. const behind = spec.at.find((a) => scenario.segments[a.segment]?.gateway); + + if (spec.egress) { + await routeScenarioExplicitly(scenario, machine, name); + log(` routed ${machine} inside the scenario, its default out through the uplink`); + continue; + } + if (!behind) { await routeViaTransit(scenario, spec, name); continue; @@ -219,6 +240,104 @@ export async function applyDefaultRoutes( } } +/** One route a machine with egress needs, so the scenario stays reachable and stays inside. */ +export interface ScenarioRoute { + /** The range this route is for. */ + cidr: string; + /** The next hop inside the scenario, or null when there is none and the range is unreachable. */ + via: string | null; +} + +/** + * The routes a machine with egress needs into the rest of the scenario. + * + * Pure, and exported, because this is the decision that keeps the uplink and the declared gateway + * from fighting — and a decision only a full raise could check is one nobody checks. + * + * One route per segment the machine is not already on, through whatever it would have defaulted to: + * its gateway if it sits behind one, transit if it sits on a public segment and transit exists. + * What has no such path is `unreachable` — the faithful translation of the state it was in before, + * where its default route pointed into scenery that dropped it, and safer, because an unreachable + * route cannot be answered by whatever network the workstation happens to sit on. + */ +export function scenarioRoutesFor(scenario: Scenario, machine: string): ScenarioRoute[] { + const spec = scenario.machines[machine]; + if (!spec || spec.at === "detached") return []; + const at = spec.at; + const onSegments = new Set(at.map((a) => a.segment)); + const behindGateway = at.some((a) => scenario.segments[a.segment]?.gateway); + + const routes: ScenarioRoute[] = []; + for (const [segment, segmentSpec] of Object.entries(scenario.segments)) { + if (onSegments.has(segment)) continue; + for (const cidr of segmentSpec.cidr) { + const slash = cidr.lastIndexOf("/"); + if (slash === -1) continue; + const v6 = cidr.slice(0, slash).includes(":"); + routes.push({ + cidr, + via: behindGateway ? gatewayInside(scenario, at, v6) : transitOn(scenario, at, v6), + }); + } + } + return routes; +} + +/** + * Apply those routes, and leave the default to the uplink. + * + * No `dev`: every next hop here is on-link, so the kernel picks the interface, and asking `awk` to + * count links is one more thing that can pick `docker0`. + */ +async function routeScenarioExplicitly( + scenario: Scenario, + machine: string, + name: string, +): Promise { + for (const { cidr, via } of scenarioRoutesFor(scenario, machine)) { + const family = cidr.slice(0, cidr.lastIndexOf("/")).includes(":") ? "-6" : "-4"; + const route = via ? `${cidr} via ${via}` : `unreachable ${cidr}`; + await incus( + ["exec", name, "--", "sh", "-c", `ip ${family} route replace ${route} 2>/dev/null || true`], + 30_000, + ); + } +} + +/** The inside address of the gateway this machine sits behind: the first host address. */ +function gatewayInside(scenario: Scenario, at: Attachment[], v6: boolean): string | null { + const behind = at.find((a) => scenario.segments[a.segment]?.gateway); + if (!behind) return null; + for (const range of scenario.segments[behind.segment]?.cidr ?? []) { + const slash = range.lastIndexOf("/"); + if (slash === -1) continue; + const base = range.slice(0, slash); + if (base.includes(":") !== v6) continue; + if (v6) return `${base.replace(/::$/, "")}::1`; + const octets = base.split("."); + octets[3] = "1"; + return octets.join("."); + } + return null; +} + +/** The transit router's address on the public segment this machine sits on. */ +function transitOn(scenario: Scenario, at: Attachment[], v6: boolean): string | null { + // One public segment means everything public is adjacent and no transit router is raised, so + // there is nothing to point at — see raiseTransit. + const publicSegments = Object.values(scenario.segments).filter((s) => s.kind === "public"); + if (publicSegments.length < 2) return null; + + const onPublic = at.find((a) => scenario.segments[a.segment]?.kind === "public"); + if (!onPublic) return null; + for (const cidr of scenario.segments[onPublic.segment]?.cidr ?? []) { + if (cidr.slice(0, cidr.lastIndexOf("/")).includes(":") !== v6) continue; + const via = transitAddress(cidr); + if (via) return via.slice(0, via.lastIndexOf("/")); + } + return null; +} + /** A machine on a public segment reaches the other public networks through transit. */ async function routeViaTransit( scenario: Scenario, diff --git a/src/lifecycle/base.ts b/src/lifecycle/base.ts index b4f169d..a9ccdd7 100644 --- a/src/lifecycle/base.ts +++ b/src/lifecycle/base.ts @@ -87,10 +87,16 @@ export async function buildBaseImage( // Trust the documentation ranges as plain-HTTP registries. // - // A scenario's registry is scenery inside the scenario, serving over HTTP, and a runtime - // will not pull from one without being told. Scoped to RFC 5737 and RFC 3849 ranges rather - // than a specific address, because those never route on the real internet — so this cannot - // make a real machine trust a real registry, whatever it is copied onto. + // **Kept after the lab's own registry was deleted, because it was never only for that.** The + // mesh HAS a registry — the `registry` module, `mesh-registry`, serving artifacts to the whole + // mesh on port 5000 over plain HTTP from whatever node runs it. In a scenario that node's + // address is a documentation-range address, and a runtime will not pull from a plain-HTTP + // registry without being told to. Take this away and the artifact store is unusable from every + // machine but the one hosting it. + // + // Scoped to RFC 5737 and RFC 3849 ranges rather than a specific address, because those never + // route on the real internet — so this cannot make a real machine trust a real registry, + // whatever it is copied onto. await incus([ "exec", BUILDER, "--", "sh", "-c", `mkdir -p /etc/docker && printf '%s' '${JSON.stringify({ @@ -162,8 +168,8 @@ export async function buildBaseImage( // Read back that the runtime will actually pull over plain HTTP from a documentation // range. Writing the file is not the same as the daemon honouring it, and a base image - // that looks right here fails much later — in a sealed scenario, as a container that - // cannot fetch its image, which is a long way from the cause. + // that looks right here fails much later — as a container that cannot fetch its image + // from the mesh's own artifact store, which is a long way from the cause. const trusted = await incusOk( ["exec", BUILDER, "--", "docker", "info", "--format", "{{.RegistryConfig.InsecureRegistryCIDRs}}"], 60_000, @@ -172,7 +178,8 @@ export async function buildBaseImage( throw new BaseImageError( `the runtime in ${BUILDER} does not trust the documentation ranges as plain-HTTP ` + `registries. It reported: ${trusted?.trim() || "nothing"}\n` + - ` Every scenario raised from this image would fail to pull from its own registry.`, + ` Every scenario raised from this image would fail to pull from the mesh's own ` + + `artifact store, which serves plain HTTP inside the scenario.`, ); } log(" trusts the documentation ranges as registries"); diff --git a/src/lifecycle/egress.ts b/src/lifecycle/egress.ts new file mode 100644 index 0000000..ae1fd1b --- /dev/null +++ b/src/lifecycle/egress.ts @@ -0,0 +1,126 @@ +/** + * Confirming a machine that says it can reach the outside actually can. + * + * **This is what the registry-reachability check became.** The old one proved that every machine + * could fetch a manifest from the registry the lab raised inside the scenario — a real check of a + * fake path, since no production mesh has such a registry. What a machine actually does is pull + * from the internet, and that is now the thing worth proving before a raise says it is finished. + * + * The failure it exists to stop is the same one, in the same shape: `raise` returns, the caller + * applies a substrate, the first pull fails, no node enrols, and the instance is left a bare + * shell — with the cause several steps back and looking like a mesh fault rather than a lab one. + * + * Two things are checked, in this order, because they fail differently and the difference is the + * whole diagnosis: + * + * - **A name resolves.** Without this the machine has a route and no way to use it, and every + * pull dies inside the runtime saying it cannot look up a host. + * - **The path carries.** A request to the registry every image ultimately comes from, over the + * uplink, through whatever gateway sits in front of this machine. Any HTTP answer counts: what + * is in question is the path, not whether Docker Hub likes us. + */ + +import type { Scenario } from "../declaration/types.ts"; +import { incus, incusOk } from "../incus/client.ts"; + +export class EgressError extends Error { + constructor(message: string) { + super(message); + this.name = "EgressError"; + } +} + +/** The host every image is fetched through, in the end. Asked for, never pulled from, here. */ +const UPSTREAM = "registry-1.docker.io"; + +/** + * Confirm every machine declaring `egress` can resolve and reach the outside. + * + * Run after the routes and the firewalls, because that is the path a pull will take: a home node's + * default is the uplink, its route to the rest of the scenario is through its gateway, and its own + * filtering is in place. Checking earlier would prove something no pull relies on. + */ +export async function confirmEgress( + scenario: Scenario, + machineNames: Map, + log: (message: string) => void = () => {}, + waitSeconds = 120, +): Promise { + for (const [machine, spec] of Object.entries(scenario.machines)) { + if (!spec.egress || spec.at === "detached") continue; + const name = machineNames.get(machine); + if (!name) continue; + + if (!(await resolves(name, waitSeconds))) { + // One repair, then a verdict. The uplink is the lab's own network and its DHCP server is + // also its resolver, so the machine has been told the answer and may simply have nowhere + // to write it — an image without systemd-resolved leaves `UseDNS=yes` inert. + await pointResolverAtTheUplink(name); + if (!(await resolves(name, 30))) { + throw new EgressError( + `${machine} declares egress and cannot resolve ${UPSTREAM}.\n` + + ` It has a route out and no way to use it, so every image pulled from the internet ` + + `would fail inside the runtime as a lookup error.\n` + + ` The uplink's DHCP server is also its resolver; this machine has not taken it.`, + ); + } + } + + const code = await reaches(name, waitSeconds); + if (!code) { + throw new EgressError( + `${machine} declares egress, resolves names, and cannot reach ${UPSTREAM}.\n` + + ` This is the PATH: its default route, the uplink, or the host's own forwarding. ` + + `Every third-party image this machine needs is pulled from the internet, so anything ` + + `applied to it would stop at the first container.`, + ); + } + log(` ${machine} reaches the internet over its uplink (${UPSTREAM} answered ${code})`); + } +} + +async function resolves(name: string, waitSeconds: number): Promise { + const deadline = Date.now() + waitSeconds * 1_000; + while (Date.now() < deadline) { + const said = await incusOk( + ["exec", name, "--", "sh", "-c", `getent hosts ${UPSTREAM} >/dev/null && echo yes`], 30_000, + ); + if (said?.trim() === "yes") return true; + await new Promise((r) => setTimeout(r, 3_000)); + } + return false; +} + +/** + * Any HTTP status at all, which is what "the path carries" means. + * + * Not 200: an unauthenticated `/v2/` is answered 401 by design, and a check demanding 200 would + * fail on a machine whose network is perfect. + */ +async function reaches(name: string, waitSeconds: number): Promise { + const deadline = Date.now() + waitSeconds * 1_000; + while (Date.now() < deadline) { + const said = (await incusOk( + ["exec", name, "--", "sh", "-c", + `curl -s -o /dev/null -w '%{http_code}' --max-time 15 https://${UPSTREAM}/v2/`], 40_000, + ))?.trim(); + if (said && /^[1-5][0-9]{2}$/.test(said)) return said; + await new Promise((r) => setTimeout(r, 5_000)); + } + return null; +} + +/** + * Write a resolver of last resort: the uplink's own gateway, which serves DHCP and DNS both. + * + * Deliberately the machine's default next hop rather than a name looked up somewhere — for a + * machine with egress that is the uplink by construction, since every scenario range is routed + * explicitly and nothing else defaults. + */ +async function pointResolverAtTheUplink(name: string): Promise { + await incus([ + "exec", name, "--", "sh", "-c", + `via=$(ip -4 route show default | awk '{print $3}' | head -n1); ` + + `[ -n "$via" ] && printf 'nameserver %s\\n' "$via" > /etc/resolv.conf; true`, + ], 30_000); +} diff --git a/src/lifecycle/place.ts b/src/lifecycle/place.ts index eb245a6..4414d12 100644 --- a/src/lifecycle/place.ts +++ b/src/lifecycle/place.ts @@ -19,6 +19,7 @@ import { join } from "node:path"; import { incus, incusOk, succeeds } from "../incus/client.ts"; import { around, log, shorten } from "../log.ts"; +import { isMeshBuilt, repositoryOf, type HeldImage } from "../pinning.ts"; /** * What this stage can put inside a machine. @@ -263,17 +264,17 @@ export async function placeImage( // sealed machine cannot reach. Measured, not assumed: the load says `Loaded image ID:` // instead of `Loaded image:`, and `docker images` then lists nothing. // - // This collides with novox/hq ADR 0006, which pins bundle images BY DIGEST and has the host - // refuse anything else. Reconciling the two needs a registry inside the scenario, which is - // real design work — see 04-ISSUES/009. + // What an archive DOES keep is the image's own ID — the digest of its configuration — and that + // is how the mesh's own images are named once loaded. See {@link loadHeldImages}. if (reference.includes("@sha256:")) { throw new PlacementError( machine, `${reference} is pinned by digest, and an image placed from an archive cannot keep its ` + `digest — a repo digest only exists for an image a registry served.\n` + ` Placing it would load an image with no name, and a container declaring that digest ` + - `would try to reach a registry the machine cannot see.\n` + - ` Place it by tag, or give the scenario a registry (novox/hq 04-ISSUES/009).`, + `would try to reach a registry.\n` + + ` Place it by tag. What survives being loaded is the image's own ID, which is what a ` + + `declaration names it by (mesh-host: an image the machine already holds).`, ); } @@ -397,3 +398,126 @@ async function waitForRuntime(instanceName: string, machine: string): Promise ({ machine, images: spec.images ?? all })) + .filter((plan) => plan.images.length > 0); +} + +/** + * Put the mesh's own images onto the machines that need them, and say what they are now called. + * + * **This is what replaced the lab's registry**, and the difference is the whole point. A registry + * inside the scenario served every image — third-party ones included — from an address that exists + * in no production mesh, so a bootstrap that could only work against it went green here and would + * have failed anywhere else. There is no such registry now: third-party images are pulled from the + * internet over each machine's `egress` uplink, and the mesh's own arrive the way they arrive on an + * operator's machine — somebody built them and put them there. + * + * The reference a declaration then uses is the image's **own ID**, the digest of its configuration. + * `docker load` preserves it, so the name is identical on the workstation that built the image and + * on every machine handed a copy — immutable, unforgeable, and requiring nothing to have served it + * (mesh-host, *an image may be named by the digest of its own configuration*). + * + * Read back on both sides. The ID is taken from the workstation and then CONFIRMED on the machine, + * because a load that lands a different image than the one exported is exactly the silent fault + * this lab exists to catch — and the reference is what every manifest will be rewritten to. + */ +export async function loadHeldImages( + scenario: Scenario, + machineNames: Map, + log: (message: string) => void = () => {}, +): Promise { + const plans = planHeldImages(scenario); + if (plans.length === 0) return []; + + const held: HeldImage[] = []; + for (const requested of scenario.images ?? []) { + // Refused by the validator, so reaching here would be a validator bug — but the consequence + // is a third-party image quietly loaded from the workstation instead of pulled, which is the + // fiction all of this exists to remove. Cheap to check, expensive to miss. + if (!isMeshBuilt(requested)) { + throw new Error( + `images: '${requested}' is not one of the mesh's own images. It is pulled from the ` + + `internet by the machine that needs it, not loaded from this workstation.`, + ); + } + + const wanted = plans.filter((plan) => plan.images.includes(requested)).map((p) => p.machine); + if (wanted.length === 0) continue; + + const id = (await local( + "docker", ["image", "inspect", "--format", "{{.Id}}", requested], 60_000, + )).stdout.trim(); + if (!/^sha256:[0-9a-f]{64}$/.test(id)) { + throw new Error( + `${requested} is not on this workstation, so there is nothing to hand the machines.\n` + + ` It is one of the mesh's own images and exists in no registry — nothing can pull it.\n` + + ` Build it first (mesh-control's \`make image …\`, or scripts/build-module-runtime.sh).`, + ); + } + const image: HeldImage = { requested, repository: repositoryOf(requested), reference: id }; + + // Exported once, handed to each machine that asked for it. The archive is the expensive part + // and it does not depend on the destination. + const tar = join(tmpdir(), `mesh-lab-held-${process.pid}-${Date.now()}.tar`); + const saved = await local("docker", ["save", requested, "-o", tar], 900_000); + if (!saved.ok) { + await unlink(tar).catch(() => {}); + throw new Error(`cannot export ${requested} from this workstation: ${saved.stderr.trim()}`); + } + + try { + for (const machine of wanted) { + const name = machineNames.get(machine); + if (!name) continue; + await waitForRuntime(name, machine); + await incus(["file", "push", tar, `${name}/tmp/held.tar`], 900_000); + const loaded = await incusOk( + ["exec", name, "--", "docker", "load", "-i", "/tmp/held.tar"], 900_000, + ); + if (!loaded?.includes("Loaded image")) { + throw new PlacementError( + machine, + `${requested} was pushed to ${machine} and did not load.\n` + + ` The runtime said: ${loaded?.trim() || "nothing"}`, + ); + } + // The reference every manifest is about to be rewritten to, confirmed present under + // exactly that name. Asking for the tag would prove the load happened; asking for the ID + // proves the thing a declaration will name is the thing that is there. + const there = (await incusOk( + ["exec", name, "--", "docker", "image", "inspect", "--format", "{{.Id}}", id], 120_000, + ))?.trim(); + if (there !== id) { + throw new PlacementError( + machine, + `${requested} loaded onto ${machine} and is not there as ${id}.\n` + + ` The runtime answered '${there || "nothing"}'.\n` + + ` Every manifest naming this image would be rewritten to a reference the machine ` + + `does not hold, and nothing serves it — so the apply would stop at the container.`, + ); + } + await succeeds(["exec", name, "--", "rm", "-f", "/tmp/held.tar"], 60_000); + } + } finally { + await unlink(tar).catch(() => {}); + } + + held.push(image); + log(` ${requested} → ${id.slice(0, 19)}… on ${wanted.join(", ")}`); + } + return held; +} diff --git a/src/lifecycle/raise.ts b/src/lifecycle/raise.ts index afe54e3..0c1aa9c 100644 --- a/src/lifecycle/raise.ts +++ b/src/lifecycle/raise.ts @@ -22,9 +22,10 @@ import { applyAddresses, applyDefaultRoutes } from "./address.ts"; import { assertSupported } from "./supported.ts"; import { planRouters, raiseRouters, raiseTransit } from "./router.ts"; import { applyHostFirewalls } from "./firewall.ts"; -import { IMAGE_PREFIX, BASE_IMAGE_ALIAS, BASE_IMAGE_HOWTO, planPlacements, applyPlacements } from "./place.ts"; +import { IMAGE_PREFIX, BASE_IMAGE_ALIAS, BASE_IMAGE_HOWTO, planPlacements, applyPlacements, loadHeldImages } from "./place.ts"; import { baseImageExists, UPSTREAM_IMAGE } from "./base.ts"; -import { confirmRegistryServes, discardStock, raiseRegistry, stockRegistry } from "./registry.ts"; +import { confirmEgress } from "./egress.ts"; +import type { HeldImage } from "../pinning.ts"; import { log as record } from "../log.ts"; /** Drivers whose snapshots are copy-on-write. On `dir` a snapshot is a full copy. */ @@ -47,12 +48,15 @@ export interface RaisedScenario { networks: string[]; pool: string; /** - * Images the scenario's registry serves, as references a declaration can pin. + * The mesh's own images, as loaded onto the machines, and what a declaration should call them. * - * Reported rather than declared, because the digest is the one this registry assigned and - * is not knowable before it was raised. + * Reported rather than declared: an image built from source has no digest until it has been + * built, and what names it here is the digest of its own configuration. + * + * **Only ours.** Everything third-party is pulled from the internet by the machine that needs + * it, so it is not in this list and nothing rewrites it. */ - images: string[]; + images: HeldImage[]; } export class RaiseError extends Error { @@ -314,24 +318,6 @@ export async function raise( const routers = await raiseRouters(scenario, instanceId, planRouters(scenario, instanceId), log); if (transit) routers.push(transit); - // Stocked on this workstation, where there is a network, and served from inside the - // scenario, where there is not (novox/hq 04-ISSUES/009). - enter("stocking the registry"); - const stock = await stockRegistry(scenario.images ?? [], log); - let registry: Awaited> = null; - try { - enter("raising the registry"); - registry = await raiseRegistry(scenario, instanceId, stock, log, pool); - } finally { - // Cleaning up scratch must not fail a raise that succeeded. The scenario is standing - // and usable; a directory left behind is untidy, and saying so is the honest report. - try { - await discardStock(stock); - } catch (err) { - log(` (could not remove the registry's scratch directory: ${(err as Error).message})`); - } - } - enter("routing machines through their gateways"); await applyDefaultRoutes(scenario, byMachine, log); @@ -340,31 +326,29 @@ export async function raise( enter("applying host firewalls"); await applyHostFirewalls(scenario, byMachine, log); - // **Only now can "the registry is serving" be said truthfully.** Raising it proved the - // registry answers on its own machine; a machine pulls across a segment, and the home nodes - // pull through a gateway whose route and firewall were applied in the two steps above. So the - // path is checked here, where it is finally the one a pull will take — and before `placing`, - // which is minutes of work that a machine unable to fetch an image cannot use. - // - // The alternative is what happened: `raise` returned, the caller applied a substrate whose - // every image is pinned to this registry, the first pull failed, no node enrolled, and the - // instance was left a bare shell. A raise that reports success owes the next step the fact it - // depends on. - if (registry) { - enter("confirming the registry serves the machines"); - await confirmRegistryServes(registry, [...byMachine.values()], stock, log); - } + // **Only now is "this machine can reach the outside" a true statement.** The route, the + // gateway and the machine's own filtering are all in place, so this is the path a pull takes. + // A raise that returned without checking would hand the next step a fact it depends on and + // has no way to test — which is how a substrate apply used to die on its first pull. + enter("confirming egress reaches the internet"); + await confirmEgress(scenario, byMachine, log); // Last, and only once the underlay is real. Placing before the machines can reach each // other would test the host against a network the scenario does not describe. enter("placing"); await applyPlacements(scenario, byMachine, log); + // After `placing`, because loading an image needs the container runtime that `placing` + // confirmed. The mesh's own images only — everything third-party is pulled by the machine + // itself, over its uplink, exactly as it is on a real one. + enter("loading the mesh's own images onto the machines"); + const images = await loadHeldImages(scenario, byMachine, log); + return { instanceId, scenario: scenario.scenario, - images: registry?.pinned ?? [], - machines: [...created, ...routers, ...(registry ? [registry.machine] : [])], + images, + machines: [...created, ...routers], networks, pool, }; diff --git a/src/lifecycle/registry.ts b/src/lifecycle/registry.ts deleted file mode 100644 index 6a6be7c..0000000 --- a/src/lifecycle/registry.ts +++ /dev/null @@ -1,487 +0,0 @@ -/** - * A registry inside the scenario. - * - * A sealed machine cannot reach a registry, and an image placed from an archive cannot keep its - * digest — `docker save` of a digest reference produces an archive with no repo tag, because a - * repo digest only exists for an image a registry served (novox/hq 04-ISSUES/009). So an image - * pinned by digest, which is the only kind the host accepts - * ([ADR 0006](../../02-DECISIONS/0046-the-installer-fetches-what-it-pins.md)), could not be - * placed at all. - * - * The answer is a registry, and it is not a workaround for the lab: ADR 0006 names an OCI - * registry as substrate, and ADR 0006 says a first node fetches "upstream, wherever the image - * ordinarily lives". **This is that upstream** — scenery, like the transit router is the - * internet ([ADR 0016](../../02-DECISIONS/0033-a-router-is-scenery-not-a-node.md)). - * - * The digests it serves are its own, not Docker Hub's, and that is correct rather than a - * compromise. What ADR 0006 requires is a reference that is exact and cannot move. A digest - * assigned by this registry is both. - */ - -import { spawn } from "node:child_process"; - -import { incus, incusOk, succeeds } from "../incus/client.ts"; -import { macFor, networkName } from "./names.ts"; -import { addressLink } from "./address.ts"; -import { around, log, shorten } from "../log.ts"; -import { BASE_IMAGE_ALIAS, placeImage } from "./place.ts"; -import { mkdtemp, rm } from "node:fs/promises"; -import { tmpdir } from "node:os"; -import { join } from "node:path"; - -/** The image the registry itself runs from. Placed by tag, which archives keep. */ -export const REGISTRY_IMAGE = "registry:2"; - -/** Where the registry serves, inside its machine. */ -export const REGISTRY_PORT = 5000; - -export class RegistryError extends Error { - constructor(message: string) { - super(message); - this.name = "RegistryError"; - } -} - -export interface StockedImage { - /** What the scenario asked for, as written. */ - requested: string; - /** The repository path the registry serves it under. */ - repository: string; - /** The digest THIS registry assigned. What a declaration pins. */ - digest: string; -} - -export interface Stock { - /** A directory holding the registry's data, ready to be placed in a machine. */ - dataDir: string; - images: StockedImage[]; -} - -/** - * Build a registry's data directory on this workstation, with the given images in it. - * - * Runs a throwaway registry here — where there IS a network — pushes into it, and keeps what - * it wrote. Research 012's reframing again: fetch at build time on a machine that has a - * network, apply on a target that needs nothing. - * - * The caller owns the returned directory and must remove it. - */ -export async function stockRegistry( - references: string[], - log: (message: string) => void = () => {}, -): Promise { - if (references.length === 0) return { dataDir: "", images: [] }; - - const dataDir = await mkdtemp(join(tmpdir(), "mesh-lab-registry-")); - const container = `mesh-lab-stock-${process.pid}`; - const port = 5000 + (process.pid % 1000); - - await docker(["rm", "-f", container], 60_000); - const started = await docker( - ["run", "-d", "--name", container, "-p", `${port}:5000`, "-v", `${dataDir}:/var/lib/registry`, - REGISTRY_IMAGE], - 300_000, - ); - if (!started.ok) { - await rm(dataDir, { recursive: true, force: true }); - throw new RegistryError( - `cannot run ${REGISTRY_IMAGE} on this workstation to stock a registry: ${started.stderr.trim()}`, - ); - } - - try { - await waitForRegistry(port); - const images: StockedImage[] = []; - - for (const reference of references) { - // The repository path a machine will pull from. A tag is dropped: what a declaration - // pins is the digest, and carrying the tag as well would invite pinning the wrong one. - const repository = repositoryFor(reference); - const target = `localhost:${port}/${repository}`; - - const tagged = await docker(["tag", reference, target], 60_000); - if (!tagged.ok) { - throw new RegistryError( - `${reference} is not on this workstation, and the lab does not fetch on a scenario's ` + - `behalf. Pull it here first.\n ${tagged.stderr.trim()}`, - ); - } - const pushed = await docker(["push", target], 900_000); - if (!pushed.ok) throw new RegistryError(`cannot push ${reference}: ${pushed.stderr.trim()}`); - - const digest = digestFrom(pushed.stdout + pushed.stderr); - if (!digest) { - throw new RegistryError( - `${reference} was pushed and the registry did not report a digest. Without one there ` + - `is nothing for a declaration to pin.`, - ); - } - images.push({ requested: reference, repository, digest }); - log(` stocked ${repository}@${digest}`); - } - - return { dataDir, images }; - } catch (err) { - await discardStock({ dataDir, images: [] }); - throw err; - } finally { - await docker(["rm", "-f", container], 60_000); - } -} - -/** - * Remove a stocked registry's data. - * - * Through a container, because a container wrote it. The registry runs as root inside, so the - * blobs it writes into a bind mount are owned by root and an ordinary process cannot remove - * them — `rmdir` fails with EACCES on a directory that looks like ours. - * - * Whoever made the files removes them. - */ -export async function discardStock(stock: Stock): Promise { - if (!stock.dataDir) return; - await docker(["run", "--rm", "-v", `${stock.dataDir}:/stock`, REGISTRY_IMAGE, - "sh", "-c", "rm -rf /stock/* /stock/.[!.]* 2>/dev/null || true"], 120_000); - await rm(stock.dataDir, { recursive: true, force: true }).catch(() => {}); -} - -/** `alpine:3.20` and `alpine` both serve from `alpine`; `foo/bar:1` from `foo/bar`. */ -export function repositoryFor(reference: string): string { - const withoutDigest = reference.split("@")[0] ?? reference; - const lastColon = withoutDigest.lastIndexOf(":"); - const lastSlash = withoutDigest.lastIndexOf("/"); - return lastColon > lastSlash ? withoutDigest.slice(0, lastColon) : withoutDigest; -} - -/** `docker push` prints `: digest: sha256:… size: …` on its last useful line. */ -export function digestFrom(output: string): string | null { - const match = output.match(/digest:\s*(sha256:[a-f0-9]{64})/); - return match?.[1] ?? null; -} - -async function waitForRegistry(port: number): Promise { - for (let i = 0; i < 30; i++) { - const probe = await docker(["run", "--rm", "--network", "host", REGISTRY_IMAGE, - "sh", "-c", `wget -q -O- http://localhost:${port}/v2/ >/dev/null 2>&1`], 30_000); - if (probe.ok) return; - await new Promise((r) => setTimeout(r, 1_000)); - } - throw new RegistryError("a registry was started on this workstation and never answered"); -} - -function docker( - args: string[], - timeoutMs: number, -): Promise<{ ok: boolean; stdout: string; stderr: string }> { - // The second of the three places the lab runs an external program (novox/hq 04-ISSUES/024). - // `docker push` of a large image is minutes of legitimate silence, which is exactly when a - // heartbeat earns its keep. - return around(`docker ${shorten(args)}`, () => runDocker(args, timeoutMs), { heartbeatMs: 15_000 }); -} - -function runDocker( - args: string[], - timeoutMs: number, -): Promise<{ ok: boolean; stdout: string; stderr: string }> { - return new Promise((resolve) => { - const child = spawn("docker", args, { stdio: ["ignore", "pipe", "pipe"] }); - let stdout = ""; - let stderr = ""; - const timer = setTimeout(() => child.kill("SIGKILL"), timeoutMs); - child.stdout.on("data", (d) => (stdout += d)); - child.stderr.on("data", (d) => (stderr += d)); - child.on("error", (err) => { - clearTimeout(timer); - resolve({ ok: false, stdout, stderr: err.message }); - }); - child.on("close", (code) => { - clearTimeout(timer); - // A docker failure is an answer here rather than an exception, so it would otherwise pass - // through the log looking exactly like a success. - if (code !== 0) { - log.debug(` exit ${code}: ${shorten([stderr.trim() || "(nothing on stderr)"], 400)}`); - } - resolve({ ok: code === 0, stdout, stderr }); - }); - }); -} - -// --- the registry inside a scenario ------------------------------------------------------------ - -/** - * Where the registry sits on its segment. - * - * A convention rather than a declaration, like the router's. `.250` is chosen to sit well away - * from the low addresses scenarios give their machines, so a scenario can be written without - * thinking about it and a collision is obvious when it happens. - */ -export const REGISTRY_HOST_OCTET = 250; - -/** The address the registry answers on, given the segment it is attached to. */ -export function registryAddress(cidr: string): string { - const [network] = cidr.split("/"); - const parts = (network ?? "").split("."); - if (parts.length !== 4) { - throw new RegistryError( - `cannot place a registry on '${cidr}': it is not an IPv4 network, and the registry needs ` + - `an address a machine can be pointed at.`, - ); - } - return `${parts[0]}.${parts[1]}.${parts[2]}.${REGISTRY_HOST_OCTET}`; -} - -/** What a declaration should pin, once a scenario is raised. */ -export function pinnedReference(address: string, image: StockedImage): string { - return `${address}:${REGISTRY_PORT}/${image.repository}@${image.digest}`; -} - -// --- raising it inside a scenario --------------------------------------------------------------- - -/** What a raised registry is, and what a declaration needs from it. */ -export interface RaisedRegistry { - machine: string; - segment: string; - address: string; - /** Each image, as a reference a declaration can pin. */ - pinned: string[]; -} - -/** - * Pick the segment the registry sits on. - * - * A public segment, because that is what stands in for the outside world — a first node fetches - * from upstream, and this is upstream. An IPv4 range, because a machine has to be pointed at it - * by address. - */ -export function registrySegment( - segments: Record, -): { name: string; cidr: string } | null { - for (const [name, segment] of Object.entries(segments)) { - if (segment.kind !== "public") continue; - const v4 = segment.cidr.find((c) => !c.includes(":")); - if (v4) return { name, cidr: v4 }; - } - return null; -} - -/** - * Raise a registry inside the scenario and load the stocked images into it. - * - * Scenery, in the same sense the transit router is: nothing under test runs on it, it holds no - * identity, and no assertion is made about its internals. It exists so that a machine can fetch - * an image the way a real one does — over the network, from a registry, by digest. - */ -export async function raiseRegistry( - scenario: { segments: Record }, - instanceId: string, - stock: Stock, - log: (message: string) => void = () => {}, - /** - * The copy-on-write pool the scenario's machines were placed on. The registry goes on it too — - * NOT the profile's default `dir` pool — so a sized root disk is thin (paid for as it fills) - * rather than a full allocation on the host's own filesystem. Absent, it falls back to the - * profile default, which is the historical behaviour for a small registry. - */ - pool?: string, -): Promise { - if (stock.images.length === 0) return null; - - const segment = registrySegment(scenario.segments); - if (!segment) { - throw new RegistryError( - `this scenario declares images and has no public IPv4 segment to serve them from.\n` + - ` The registry stands in for the outside world, so it sits on a public segment.`, - ); - } - - const address = registryAddress(segment.cidr); - const name = `mlab-${instanceId}-registry`; - const prefix = segment.cidr.slice(segment.cidr.lastIndexOf("/")); - - if (!(await succeeds(["config", "show", name], 15_000))) { - // The registry holds the WHOLE `images:` union on its own root disk, and the base image's - // default is only ~10GiB. A single-node bed stocks a handful of images and fits; a broad bed - // — and especially the full segmented mesh, whose union is both server sets at once (~28GiB) - // — overflows it, and the raise dies "no space left on device" while pushing blobs into the - // registry. So the registry gets a sized root disk. Thin on a copy-on-write pool, so a small - // bed pays only for what it actually stocks; MESH_LAB_REGISTRY_DISK overrides for a giant one. - const registryDisk = process.env["MESH_LAB_REGISTRY_DISK"] ?? "80GiB"; - await incus([ - "init", BASE_IMAGE_ALIAS, name, "--vm", - "-c", "security.secureboot=false", - "-c", "limits.memory=1GiB", - ...(pool ? ["-s", pool] : []), - "-d", `root,size=${registryDisk}`, - "-c", `user.mesh-lab.instance=${instanceId}`, - // Tagged as a machine as well, so `destroy` finds it with one query — a router that - // carried only its own tag was left behind and held its networks open. - "-c", "user.mesh-lab.machine=registry", - "-c", "user.mesh-lab.registry=true", - ], 300_000); - await succeeds(["config", "device", "remove", name, "eth0"], 15_000); - await incus([ - "config", "device", "add", name, "eth0", "nic", - "nictype=bridged", - `parent=${networkName(instanceId, segment.name)}`, - `hwaddr=${macFor(instanceId, "registry", 0)}`, - ]); - } - await succeeds(["start", name], 60_000); - await waitForAgent(name); - - // Addressed the way every other machine is: a systemd-networkd unit matching the MAC. - // - // **This used to be `ip addr add`, and it stalled the lab.** An address set by hand leaves - // networkd waiting to configure a link it was never told about, so the link sits at - // `configuring`, `systemd-networkd-wait-online` never returns — its timeout is `infinity` — - // and `network-online.target` is never reached. Docker is ordered after that target, so - // `docker load` two lines below blocked on a socket whose daemon was queued behind a target - // that would never come. - // - // Matching on MAC and not on interface name is still the rule: a machine with a container - // runtime has a `docker0` that sorts before `enp5s0`, and naive selection configures that. - await addressLink(name, { - device: "eth0", - mac: macFor(instanceId, "registry", 0), - addresses: [`${address}${prefix}`], - // The registry takes the segment's default. It carried no MTU before this and still does - // not: what a scenario sets an MTU for is the path under test, and this is scenery. - mtu: undefined, - }); - - log(` registry on ${segment.name} at ${address}`); - - // The registry's own image, placed by tag — an archive keeps a tag and cannot keep a digest, - // which is the whole reason this machine exists. - // Logged, not silenced. This is the step a stall sat in for thirty-five minutes while the - // caller had passed it a callback that threw everything away (novox/hq 04-ISSUES/024). - await placeImage(name, "registry", REGISTRY_IMAGE, log); - - // The destination must EXIST before a recursive push, or incus copies the source's contents - // rather than the source — the data lands one directory too shallow, the registry finds - // nothing where it looks, and every pull fails with `not found`. - await incus(["exec", name, "--", "mkdir", "-p", "/srv/registry"], 60_000); - await incus(["file", "push", "-r", `${stock.dataDir}/docker`, `${name}/srv/registry/`], 900_000); - - await incus(["exec", name, "--", "docker", "run", "-d", - "--name", "registry", "--restart", "unless-stopped", - "-p", `${REGISTRY_PORT}:5000`, - "-v", "/srv/registry:/var/lib/registry", - REGISTRY_IMAGE], 300_000); - - // Read back that each image is SERVED, by asking for its manifest by digest — which is - // exactly what a machine will do. - // - // Not that the catalog endpoint answers: `{"repositories":[]}` contains the word - // `repositories`, so checking for that passed on a registry holding nothing at all, and the - // failure surfaced much later as a container that could not be pulled. - let answered = false; - for (let i = 0; i < 20 && !answered; i++) { - const ping = await incusOk(["exec", name, "--", "curl", "-s", "-o", "/dev/null", - "-w", "%{http_code}", "--max-time", "3", - `http://localhost:${REGISTRY_PORT}/v2/`], 30_000); - answered = ping?.trim() === "200"; - if (!answered) await new Promise((r) => setTimeout(r, 2_000)); - } - if (!answered) { - throw new RegistryError( - `the registry on ${name} started and never answered. Machines in this scenario cannot ` + - `fetch an image, so nothing that declares a container will work.`, - ); - } - - const pinned: string[] = []; - for (const image of stock.images) { - const code = await incusOk(["exec", name, "--", "curl", "-s", "-o", "/dev/null", - "-w", "%{http_code}", "--max-time", "5", - "-H", "Accept: application/vnd.docker.distribution.manifest.v2+json", - `http://localhost:${REGISTRY_PORT}/v2/${image.repository}/manifests/${image.digest}`, - ], 60_000); - if (code?.trim() !== "200") { - throw new RegistryError( - `the registry on ${name} is running and does not serve ${image.repository}@${image.digest} ` + - `(it answered ${code?.trim() || "nothing"}).\n` + - ` The images were stocked on this workstation and did not arrive intact, so a ` + - `machine declaring that image would fail to pull it.`, - ); - } - const reference = pinnedReference(address, image); - pinned.push(reference); - log(` serving ${reference}`); - } - return { machine: name, segment: segment.name, address, pinned }; -} - -/** - * Confirm the registry serves the MACHINES, not just itself. - * - * **`raiseRegistry` proves the wrong thing, and the difference cost a whole raise.** It curls - * `localhost:5000` from inside the registry's own machine — which says the registry process is up - * and holds the blobs, and says nothing at all about the path every other machine actually uses: - * across a segment, and for the home nodes through a NAT gateway whose route is applied two steps - * LATER. So `raise` could return "serving" with the anchor unable to reach the registry at all, the - * substrate apply's first pull would fail, no node could enrol, and the instance was left a bare - * shell — VMs and a registry and nothing else. - * - * A raise is not finished while that is still possible. This is the check that makes "the registry - * is serving" mean what the next step needs it to mean: from each machine, on the address it will - * pin, over the network it will use, once its route and its firewall are in place. - * - * **What it proves and what it does not.** It asks for `/v2/` and then for one stocked manifest BY - * DIGEST — the same two requests a pull begins with, from the same place. It does not fetch layers: - * every digest was already read back inside the registry machine, so what is in question here is - * the path, not the content, and a probe per machine keeps the check to seconds rather than the - * many minutes a full pull of a hundred images would take. - */ -export async function confirmRegistryServes( - registry: RaisedRegistry, - machines: string[], - stock: Stock, - log: (message: string) => void = () => {}, - waitSeconds = 180, -): Promise { - const probe = stock.images[0]; - if (!probe) return; - const base = `http://${registry.address}:${REGISTRY_PORT}`; - const manifest = - `-H "Accept: application/vnd.docker.distribution.manifest.v2+json" ` + - `${base}/v2/${probe.repository}/manifests/${probe.digest}`; - - for (const machine of machines) { - const deadline = Date.now() + waitSeconds * 1_000; - let last = ""; - let served = false; - while (!served && Date.now() < deadline) { - // One shell, two requests: the machine can reach the registry AND the registry answers for - // an image by the digest a declaration pins. Either alone passes on a registry serving - // nothing, which is the failure this whole function exists to stop reporting as success. - const said = await incusOk(["exec", machine, "--", "sh", "-c", - `printf '%s %s' ` + - `"$(curl -s -o /dev/null -w '%{http_code}' --max-time 5 ${base}/v2/)" ` + - `"$(curl -s -o /dev/null -w '%{http_code}' --max-time 10 ${manifest})"`, - ], 40_000); - last = said?.trim() ?? ""; - served = last === "200 200"; - if (!served) await new Promise((r) => setTimeout(r, 3_000)); - } - if (!served) { - throw new RegistryError( - `${machine} cannot pull from the registry at ${registry.address}:${REGISTRY_PORT} after ` + - `${waitSeconds}s (it got "${last || "nothing"}" for /v2/ and for ` + - `${probe.repository}@${probe.digest}).\n` + - ` The registry answers on its own machine, so this is the PATH: this machine's route, ` + - `its gateway, or its firewall. Every image this scenario declares is unreachable from ` + - `here, so anything applied to it would fail on its first pull.`, - ); - } - log(` ${machine} can pull from ${registry.address}:${REGISTRY_PORT}`); - } -} - -async function waitForAgent(name: string): Promise { - for (let i = 0; i < 90; i++) { - if (await succeeds(["exec", name, "--", "true"], 10_000)) return; - await new Promise((r) => setTimeout(r, 2_000)); - } - throw new RegistryError(`${name} started and its agent never answered.`); -} diff --git a/src/pinning.ts b/src/pinning.ts index 110f2e1..9da9752 100644 --- a/src/pinning.ts +++ b/src/pinning.ts @@ -1,55 +1,101 @@ /** - * Rewriting an image reference to the one a scenario's own registry serves. + * Naming the mesh's own images by what they are. * * **A digest is not knowable until something is built** (novox/hq 04-ISSUES/025). A manifest in a * repository can pin a third-party image, because somebody can ask a registry what a tag points - * at. It cannot pin an image the mesh builds itself: that image does not exist yet, and when it - * does its digest belongs to whichever registry served it. + * at. It cannot pin an image the mesh builds itself: mesh-control, mesh-builder, mesh-route-proxy, + * the per-module runtimes and the provisioners exist in no registry, so there is no manifest + * digest to write down. The catalogue ships sixty-four zeros for them, which parses, resolves, + * composes — and stops on the machine. * - * The bundle has always had this problem and solves it by rewriting references once the scenario's - * registry is up and its digests are known. Modules have exactly the same problem and were solving - * it by shipping sixty-four zeros, which parses, resolves, composes — and stops on the machine. + * The lab used to answer that with a registry of its own: raise one inside the scenario, push + * everything into it, and rewrite every reference — third-party ones included — to the digest it + * assigned. **That registry does not exist in production, so the lab was testing a fiction**, and + * the fiction hid the bootstrap problems it was supposed to find. * - * So the rewriting is shared rather than copied, and matches on the **repository**, because that - * is the part a person writes and the only part that survives being served somewhere else. + * What is true instead is two things: + * + * - **Third-party images are pulled from the internet.** They are left exactly as written, and + * the machine fetches them over its `egress` uplink the way any machine does. + * - **The mesh's own images are built and handed over.** They are loaded onto the machine from + * the workstation that built them, and named by the digest of their own image configuration — + * a bare `sha256:…`, which mesh-host accepts as "an image this machine already holds" + * (mesh-host, *an image may be named by the digest of its own configuration*). + * + * So the rewriting that remains is only the second kind, and it matches on the **repository**, + * because that is the part a person writes and the only part a placeholder digest does not say. */ -/** `192.0.2.250:5000/ghcr.io/mailu/admin@sha256:…` → `ghcr.io/mailu/admin` */ -export function repositoryOf(pinned: string): string { - const at = pinned.indexOf("@"); - const body = at === -1 ? pinned : pinned.slice(0, at); - const slash = body.indexOf("/"); - // Everything after the registry. A reference with no slash at all is its own repository. - return slash === -1 ? body : body.slice(slash + 1); +/** What the lab loaded onto a machine, and what a declaration should call it. */ +export interface HeldImage { + /** As the scenario asked for it, a tag this workstation holds: `mesh-runtime-postgres:development`. */ + requested: string; + /** What a manifest names it by, with no tag and no digest: `mesh-runtime-postgres`. */ + repository: string; + /** + * The reference a declaration uses: a bare `sha256:<64 hex>`. + * + * The image's own ID — the digest of its configuration — which `docker load` preserves, so the + * name is the same on the workstation that built it and on every machine it was handed to. + */ + reference: string; +} + +/** `alpine:3.20` and `alpine` both mean `alpine`; `foo/bar:1` means `foo/bar`. */ +export function repositoryOf(reference: string): string { + const withoutDigest = reference.split("@")[0] ?? reference; + const lastColon = withoutDigest.lastIndexOf(":"); + const lastSlash = withoutDigest.lastIndexOf("/"); + return lastColon > lastSlash ? withoutDigest.slice(0, lastColon) : withoutDigest; } /** - * Replace every reference to a stocked repository with the reference this scenario serves. + * Whether a reference names an image the mesh builds for itself. * - * Matching is on the repository and ignores whatever registry and digest were written down — - * a file may name `postgres@sha256:7456…` or `mesh-provision-postgres@sha256:0000…` and both mean - * *the postgres this scenario has*. That is the whole point: the text says which image, the - * scenario says which copy. + * **Derived from the shape the build produces, not from a list of names.** `make image + * builder-image provisioner-image objectstore-image redis-provisioner-image proxy-image` in + * mesh-control and `scripts/build-module-runtime.sh` here both tag their output `mesh-` + * with no registry host and no upstream organisation — that is what "built here, published + * nowhere" looks like, and a hardcoded list would go stale the first time a module is added. * - * A repository the scenario did not stock is left alone rather than blanked. It may be reachable - * some other way, and silently emptying a reference would produce the exact failure this exists to - * prevent. + * The absence of a slash carries the weight: `ghcr.io/mailu/admin` and + * `registry.example/novox/www` both say where they are fetched from, and a bare + * `mesh-runtime-plex` says there is nowhere. */ -export function pinnedInto(text: string, served: string[]): string { +export function isMeshBuilt(reference: string): boolean { + const repository = repositoryOf(reference); + return !repository.includes("/") && repository.startsWith("mesh-"); +} + +/** + * Replace every reference to one of the mesh's own images with the image the machine holds. + * + * Matching is on the repository and ignores whatever digest was written down — a manifest says + * `mesh-runtime-postgres@sha256:0000…` and means *the runtime this machine was given*. + * + * **Everything else is left exactly as it is.** `postgres@sha256:7456…`, `gitea/gitea@sha256:…` + * and `ghcr.io/mailu/admin@sha256:…` are pulled from the internet over the machine's uplink, which + * is what a real machine does and the reason the lab's own registry is gone. + */ +export function pinnedInto(text: string, held: HeldImage[]): string { let out = text; - for (const pinned of served) { - const repository = repositoryOf(pinned); - const escaped = repository.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); + for (const image of held) { + const escaped = image.repository.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); // Optionally a registry, then the repository, then any digest. Anchored on a quote or // whitespace so a longer repository ending in a shorter one is not half-replaced. out = out.replaceAll( new RegExp(`(?<=^|["\\s])(?:[A-Za-z0-9_.:-]+\\/)*${escaped}@sha256:[0-9a-f]{64}`, "g"), - pinned, + image.reference, ); } return out; } +/** The reference for one repository, or undefined if this scenario loaded no such image. */ +export function referenceFor(held: HeldImage[], repository: string): string | undefined { + return held.find((image) => image.repository === repository)?.reference; +} + /** Whether anything is still pinned to a placeholder, which would fail on the machine. */ export function stillUnpinned(text: string): string[] { return [...text.matchAll(/([A-Za-z0-9_.:/-]+)@sha256:0{64}/g)].map((m) => m[1]!); diff --git a/src/suite.ts b/src/suite.ts index 8496026..47682be 100644 --- a/src/suite.ts +++ b/src/suite.ts @@ -53,9 +53,9 @@ export async function runSuite(args: string[]): Promise { // the long run reaches the end of that path in about 160 seconds. // // So it cost a whole scenario, every passing run, to save about 45 seconds on a failing one. - // A scenario is three machines including a registry that boots a kernel to serve files, which - // is where the two minutes went. `test/integration/canary.test.ts` is still there and still - // runs when it is named; it is no longer raised on the way to everything else. + // A scenario is machines that each boot a kernel, which is where the two minutes went. + // `test/integration/canary.test.ts` is still there and still runs when it is named; it is no + // longer raised on the way to everything else. const { code, seen } = await runFiles(files); console.log("\n" + reportOn(counted(seen), (p, f) => record(p, f, files, process.env, against))); diff --git a/src/warm.ts b/src/warm.ts index f8762d0..902a243 100644 --- a/src/warm.ts +++ b/src/warm.ts @@ -23,6 +23,7 @@ import { homedir } from "node:os"; import { list, restore, snapshot, snapshots, destroy } from "./lifecycle/operate.ts"; import type { Against } from "./lastrun.ts"; import { whatWasTested } from "./lastrun.ts"; +import type { HeldImage } from "./pinning.ts"; /** The state a warm instance is kept at. One label, because a second is a state nobody named. */ export const label = "warm"; @@ -31,13 +32,13 @@ export interface Warm { scenario: string; instanceId: string; /** - * The image references the scenario's registry serves, pinned by digest. + * The mesh's own images, as loaded onto this instance's machines, by the ID each is held under. * * Kept because they are worked out while raising and a restored instance never raises. Without - * them a warm run knows nothing about what it can pull, and every test naming an image fails - * for a reason that has nothing to do with what it was testing. + * them a warm run knows nothing about what its machines hold, and every test naming one of our + * images fails for a reason that has nothing to do with what it was testing. */ - images: string[]; + images: HeldImage[]; /** The commit each repository was at when this was brought to its state. */ against: Against; at: string; @@ -187,23 +188,23 @@ export async function cool(): Promise { } /** - * What a raised scenario stocked, held until it is kept. + * What a raised scenario loaded onto its machines, held until it is kept. * * Raising works the images out and snapshotting happens later, so this carries them between the * two without the caller having to hold them. */ -const stock = new Map(); +const stock = new Map(); -export function rememberStock(instanceId: string, images: string[]): void { +export function rememberStock(instanceId: string, images: HeldImage[]): void { stock.set(instanceId, images); } -function stockOf(instanceId: string): string[] { +function stockOf(instanceId: string): HeldImage[] { return stock.get(instanceId) ?? []; } -/** What a restored instance's registry serves, from when it was warmed. */ -export function warmStock(instanceId: string): { images: string[] } { +/** What a restored instance's machines hold, from when it was warmed. */ +export function warmStock(instanceId: string): { images: HeldImage[] } { const warm = remembered(); if (!warm || warm.instanceId !== instanceId) return { images: [] }; return { images: warm.images };