One name per thing, per the HQ glossary: the module/container/image/binary/repo becomes mesh-controller, the seat the-controller, and the store+broker pair the foundation (embedded base bundles, default template and example lock renamed with their go:embed directives). No behaviour change — a pure vocabulary rename. Claude-Session: https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx
209 lines
10 KiB
TypeScript
209 lines
10 KiB
TypeScript
/**
|
|
* Confirming a machine that says it can reach the outside actually can.
|
|
*
|
|
* **This is what the registry-reachability check became.** The old one proved that every machine
|
|
* could fetch a manifest from the registry the lab raised inside the scenario — a real check of a
|
|
* fake path, since no production mesh has such a registry. What a machine actually does is pull
|
|
* from the internet, and that is now the thing worth proving before a raise says it is finished.
|
|
*
|
|
* The failure it exists to stop is the same one, in the same shape: `raise` returns, the caller
|
|
* applies a foundation, the first pull fails, no node enrols, and the instance is left a bare
|
|
* shell — with the cause several steps back and looking like a mesh fault rather than a lab one.
|
|
*
|
|
* Two things are checked, in this order, because they fail differently and the difference is the
|
|
* whole diagnosis:
|
|
*
|
|
* - **A name resolves.** Without this the machine has a route and no way to use it, and every
|
|
* pull dies inside the runtime saying it cannot look up a host.
|
|
* - **The path carries.** A request to the registry every image ultimately comes from, over the
|
|
* uplink, through whatever gateway sits in front of this machine. Any HTTP answer counts: what
|
|
* is in question is the path, not whether Docker Hub likes us.
|
|
*/
|
|
|
|
import type { Scenario } from "../declaration/types.ts";
|
|
import { incus, incusOk } from "../incus/client.ts";
|
|
|
|
export class EgressError extends Error {
|
|
constructor(message: string) {
|
|
super(message);
|
|
this.name = "EgressError";
|
|
}
|
|
}
|
|
|
|
/** The host every image is fetched through, in the end. Asked for, never pulled from, here. */
|
|
const UPSTREAM = "registry-1.docker.io";
|
|
|
|
/**
|
|
* Confirm every machine declaring `egress` can resolve and reach the outside.
|
|
*
|
|
* Run after the routes and the firewalls, because that is the path a pull will take: a home node's
|
|
* default is the uplink, its route to the rest of the scenario is through its gateway, and its own
|
|
* filtering is in place. Checking earlier would prove something no pull relies on.
|
|
*/
|
|
export async function confirmEgress(
|
|
scenario: Scenario,
|
|
machineNames: Map<string, string>,
|
|
log: (message: string) => void = () => {},
|
|
waitSeconds = 120,
|
|
): Promise<void> {
|
|
for (const [machine, spec] of Object.entries(scenario.machines)) {
|
|
if (!spec.egress || spec.at === "detached") continue;
|
|
const name = machineNames.get(machine);
|
|
if (!name) continue;
|
|
|
|
if (!(await resolves(name, waitSeconds))) {
|
|
// One repair, then a verdict. The uplink is the lab's own network and its DHCP server is
|
|
// also its resolver, so the machine has been told the answer and may simply have nowhere
|
|
// to write it — an image without systemd-resolved leaves `UseDNS=yes` inert.
|
|
await pointResolverAtTheUplink(name);
|
|
if (!(await resolves(name, 30))) {
|
|
throw new EgressError(
|
|
`${machine} declares egress and cannot resolve ${UPSTREAM}.\n` +
|
|
` It has a route out and no way to use it, so every image pulled from the internet ` +
|
|
`would fail inside the runtime as a lookup error.\n` +
|
|
` The uplink's DHCP server is also its resolver; this machine has not taken it.`,
|
|
);
|
|
}
|
|
}
|
|
|
|
const code = await reaches(name, waitSeconds);
|
|
if (!code) {
|
|
throw new EgressError(
|
|
`${machine} declares egress, resolves names, and cannot reach ${UPSTREAM}.\n` +
|
|
` This is the PATH: its default route, the uplink, or the host's own forwarding. ` +
|
|
`Every third-party image this machine needs is pulled from the internet, so anything ` +
|
|
`applied to it would stop at the first container.`,
|
|
);
|
|
}
|
|
log(` ${machine} reaches the internet over its uplink (${UPSTREAM} answered ${code})`);
|
|
|
|
// **Now that the path is proven, stop one dropped packet from costing a whole run.**
|
|
//
|
|
// The check above deliberately uses the uplink alone, because proving that path is the point
|
|
// of it. What follows is a different concern: a bed runs for hours after this, pulling images
|
|
// on four machines at once, and the uplink's resolver is a single server under exactly that
|
|
// load. Three runs have died at a lookup timeout well after this check passed — not because
|
|
// the path was broken, but because one query went unanswered.
|
|
//
|
|
// So the uplink stays first and keeps being used; public resolvers are appended behind it, and
|
|
// the retry is tightened so a silent server costs seconds rather than the step. A broken uplink
|
|
// still fails the check above, before any of this.
|
|
await resilientResolver(name);
|
|
|
|
// **And then prove it is STEADY, because one success proved nothing.**
|
|
//
|
|
// The check above is satisfied by a single answer, and that is how a machine resolving one
|
|
// query in three passed it and then killed two installs twenty minutes later. What a run needs
|
|
// is not "a name resolved once" but "names resolve reliably", and those differ by exactly the
|
|
// failure that has cost the most time here. Consecutive, because alternating success and
|
|
// timeout is the observed shape — a total count would pass on the same machine.
|
|
const steady = await steadilyResolves(name, 5);
|
|
if (steady < 5) {
|
|
throw new EgressError(
|
|
`${machine} resolves ${UPSTREAM} only ${steady} time(s) in five consecutive tries.\n` +
|
|
` It has egress and an unreliable resolver, which does not fail here — it fails later, ` +
|
|
`inside a pull or a clone, as "could not resolve host" with the cause long out of view.\n` +
|
|
` The uplink's own resolver is the usual culprit and is deliberately last in the list; ` +
|
|
`a machine still failing this has something wrong upstream of the lab.`,
|
|
);
|
|
}
|
|
log(` ${machine} resolves steadily (5/5)`);
|
|
}
|
|
}
|
|
|
|
/** How many of `tries` consecutive lookups answered. Stops at the first failure. */
|
|
async function steadilyResolves(name: string, tries: number): Promise<number> {
|
|
for (let i = 0; i < tries; i++) {
|
|
const said = await incusOk(
|
|
["exec", name, "--", "sh", "-c",
|
|
`timeout 8 getent hosts ${UPSTREAM} >/dev/null && echo yes`], 20_000,
|
|
);
|
|
if (said?.trim() !== "yes") return i;
|
|
}
|
|
return tries;
|
|
}
|
|
|
|
async function resolves(name: string, waitSeconds: number): Promise<boolean> {
|
|
const deadline = Date.now() + waitSeconds * 1_000;
|
|
while (Date.now() < deadline) {
|
|
const said = await incusOk(
|
|
["exec", name, "--", "sh", "-c", `getent hosts ${UPSTREAM} >/dev/null && echo yes`], 30_000,
|
|
);
|
|
if (said?.trim() === "yes") return true;
|
|
await new Promise((r) => setTimeout(r, 3_000));
|
|
}
|
|
return false;
|
|
}
|
|
|
|
/**
|
|
* Any HTTP status at all, which is what "the path carries" means.
|
|
*
|
|
* Not 200: an unauthenticated `/v2/` is answered 401 by design, and a check demanding 200 would
|
|
* fail on a machine whose network is perfect.
|
|
*/
|
|
async function reaches(name: string, waitSeconds: number): Promise<string | null> {
|
|
const deadline = Date.now() + waitSeconds * 1_000;
|
|
while (Date.now() < deadline) {
|
|
const said = (await incusOk(
|
|
["exec", name, "--", "sh", "-c",
|
|
`curl -s -o /dev/null -w '%{http_code}' --max-time 15 https://${UPSTREAM}/v2/`], 40_000,
|
|
))?.trim();
|
|
if (said && /^[1-5][0-9]{2}$/.test(said)) return said;
|
|
await new Promise((r) => setTimeout(r, 5_000));
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/**
|
|
* Write a resolver of last resort: the uplink's own gateway, which serves DHCP and DNS both.
|
|
*
|
|
* Deliberately the machine's default next hop rather than a name looked up somewhere — for a
|
|
* machine with egress that is the uplink by construction, since every scenario range is routed
|
|
* explicitly and nothing else defaults.
|
|
*/
|
|
/**
|
|
* Give the machine's uplink more than one resolver, through the thing that owns resolvers.
|
|
*
|
|
* **Not by writing /etc/resolv.conf**, which was the first attempt and was wrong: on these images
|
|
* that path is a symlink managed by systemd-resolved, so a file written over it is either reverted
|
|
* or breaks the link. Checked on a running machine rather than assumed.
|
|
*
|
|
* The shape of the problem is visible in `resolvectl status`: the machine has sensible global
|
|
* fallbacks, and the *link* carrying the default route has exactly one server — the uplink gateway.
|
|
* resolved will not reach for a global fallback while the link it is using has a server of its own,
|
|
* so one unanswered packet is one failed lookup.
|
|
*
|
|
* **The uplink goes last, and this is a preference, not a fix.** The uplink's resolver is a single
|
|
* dnsmasq with no unique knowledge any scenario machine needs — machines here address each other by
|
|
* address, and the mesh writes its own names into /etc/hosts — so there is no reason for anything
|
|
* to wait on it. It is kept, last, so DHCP-supplied names still answer.
|
|
*
|
|
* **It is written down as a preference because it was once mistaken for the cure.** Lookups were
|
|
* timing out two times in three; the uplink was first in the list; reordering it made five in five;
|
|
* the conclusion drew itself and was wrong. The machines were sharing ONE DHCP lease (see
|
|
* `distinguishMachines` in raise.ts), so which machine could resolve anything depended on which had
|
|
* last won an ARP race — and reordering resolvers on a machine that has just won looks exactly like
|
|
* a fix. Nothing below this line will save a bed whose machines share an address.
|
|
*
|
|
* The caches are flushed afterwards, because resolved remembers the failures it collected while
|
|
* the bad server was in front.
|
|
*/
|
|
async function resilientResolver(name: string): Promise<void> {
|
|
await incus([
|
|
"exec", name, "--", "sh", "-c",
|
|
`link=$(ip -4 route show default | awk '{print $5}' | head -n1); ` +
|
|
`via=$(ip -4 route show default | awk '{print $3}' | head -n1); ` +
|
|
`if [ -n "$link" ] && command -v resolvectl >/dev/null 2>&1; then ` +
|
|
`resolvectl dns "$link" 1.1.1.1 8.8.8.8 9.9.9.9 $via >/dev/null 2>&1 || true; ` +
|
|
`resolvectl flush-caches >/dev/null 2>&1 || true; fi; true`,
|
|
], 30_000);
|
|
}
|
|
|
|
async function pointResolverAtTheUplink(name: string): Promise<void> {
|
|
await incus([
|
|
"exec", name, "--", "sh", "-c",
|
|
`via=$(ip -4 route show default | awk '{print $3}' | head -n1); ` +
|
|
`[ -n "$via" ] && printf 'nameserver %s\\n' "$via" > /etc/resolv.conf; true`,
|
|
], 30_000);
|
|
}
|