A network-checker module: dial what the mesh claims, from where the callers are
The mesh asserts three things are callable (ADR 0144) — what runs on the same machine, another machine's service exposed to the private network, and another machine's service exposed publicly — and has never checked any of them. The first was broken for eleven hours while the mesh reported every machine healthy. This runs on every machine, on the cadence the mesh already has, in its own container: the same position every other module calls from. Not the host and not the control plane, both of which reach these addresses by paths no ordinary caller uses and would have passed throughout that outage. **Its probe is its own endpoint, and that is the point.** Declared reachable over the private network like any other service, so it is admitted by exactly the rule that governs every internally-exposed service and fails when that rule is wrong. The tempting target is a service every machine has, and those are the ones never closed — ssh above all — which would have passed while the thing that actually broke was a service exposed to the private network. It resolves before it dials and says which failed, because a name that does not resolve and a port that does not answer have different owners. One failure is not a fault: a machine rebooting is ordinary, so a path is broken after consecutive runs and the count travels with the result. It reports and repairs nothing. novox/hq ADR 0145. Eight tests; the consecutive-failure logic proved by reverting it once. Not yet registered or assigned.
This commit is contained in:
@@ -0,0 +1,84 @@
|
||||
// Dial everything the mesh claims is reachable, and say what was found (novox/hq ADR 0145).
|
||||
//
|
||||
// Runs on a cadence, from this machine, in this module's own container — the same position every other
|
||||
// module on the machine calls from. That is the whole point: a check run by the host or by the control
|
||||
// plane reaches these addresses by a path no ordinary caller uses, and would have passed throughout the
|
||||
// outage that produced this module (novox/hq 04-ISSUES/145).
|
||||
//
|
||||
// It reports and does nothing else. A checker that repaired things would be a second control plane.
|
||||
|
||||
import { readFileSync, writeFileSync, mkdirSync, renameSync } from "node:fs";
|
||||
import { dirname, join } from "node:path";
|
||||
|
||||
import { dial, tally, targetsFor, type Counts, type Result, type Roster } from "../reach.js";
|
||||
|
||||
/** Where the mesh renders this machine's view of the others, and where the counts are kept between runs. */
|
||||
const rosterFile = process.env.MESH_NETWORK_CHECKER_ROSTER ?? "/run/config/roster.json";
|
||||
const stateDir = process.env.MESH_NETWORK_CHECKER_STATE ?? "/run/state";
|
||||
const probePort = Number(process.env.MESH_NETWORK_CHECKER_PORT ?? "9876");
|
||||
const publicPort = process.env.MESH_NETWORK_CHECKER_PUBLIC_PORT
|
||||
? Number(process.env.MESH_NETWORK_CHECKER_PUBLIC_PORT)
|
||||
: undefined;
|
||||
const timeoutMs = Number(process.env.MESH_NETWORK_CHECKER_TIMEOUT_MS ?? "4000");
|
||||
const threshold = Number(process.env.MESH_NETWORK_CHECKER_THRESHOLD ?? "2");
|
||||
|
||||
/** read is a JSON file or a stated failure — never a silent default, which is how a checker comes to
|
||||
* report that everything is fine because it read nothing. */
|
||||
function read<T>(path: string, whenMissing: T | null): T {
|
||||
try {
|
||||
return JSON.parse(readFileSync(path, "utf8")) as T;
|
||||
} catch (err) {
|
||||
if (whenMissing !== null) return whenMissing;
|
||||
console.error(`network-checker: cannot read ${path}: ${(err as Error).message}`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
function writeAtomically(path: string, body: string): void {
|
||||
mkdirSync(dirname(path), { recursive: true });
|
||||
const temp = `${path}.writing`;
|
||||
writeFileSync(temp, body);
|
||||
renameSync(temp, path);
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const roster = read<Roster>(rosterFile, null);
|
||||
if (!roster.machines?.length) {
|
||||
console.error("network-checker: the roster names no machines; nothing to check");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const targets = targetsFor(roster, probePort, publicPort);
|
||||
// In parallel, because a machine that is away should not delay the rest: a run that takes
|
||||
// machines × timeout would outlast its own cadence on a mesh of any size.
|
||||
const results: Result[] = await Promise.all(targets.map((t) => dial(t, timeoutMs)));
|
||||
|
||||
const countsFile = join(stateDir, "consecutive.json");
|
||||
const { counts, broken } = tally(results, read<Counts>(countsFile, {}), threshold);
|
||||
writeAtomically(countsFile, JSON.stringify(counts, null, 1));
|
||||
|
||||
// Written whole, every run: a reader asking "what does this machine reach" gets an answer about now
|
||||
// rather than the last time something changed.
|
||||
writeAtomically(join(stateDir, "reach.json"), JSON.stringify({
|
||||
node: roster.node,
|
||||
at: new Date().toISOString(),
|
||||
checked: results.length,
|
||||
broken: broken.length,
|
||||
results,
|
||||
}, null, 1));
|
||||
|
||||
for (const b of broken) {
|
||||
console.error(
|
||||
`network-checker: ${roster.node} cannot reach ${b.machine} (${b.claim}) at ${b.at}:${b.port} — ` +
|
||||
`${b.failed} failed${b.detail ? `: ${b.detail}` : ""}, ${b.consecutive} run(s) running`);
|
||||
}
|
||||
if (broken.length === 0) {
|
||||
console.log(`network-checker: ${roster.node} reaches all ${results.length} checked path(s)`);
|
||||
}
|
||||
|
||||
// A broken path is not this process failing. It did its job; exiting non-zero would make the mesh
|
||||
// read the checker as the fault, and a scheduled step that fails is retried rather than believed.
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
void main();
|
||||
Reference in New Issue
Block a user