Files
mesh-host/packaging/launch_test.sh
T
jschoubben fbf0fb7d63 The host delivers its own successor, and versions live side by side
The supervision was already right: a clean exit means the host stood aside, and
the launcher's next turn runs what is on disk. Two things made it dead code —
nothing told the running host a successor was waiting, and the rollback resolved
its known-good version through pacman, which no machine here uses and which two
of three operating systems do not have.

Keeping a version rather than a path was the clue. Versions now live in
directories named for them:

- the launcher picks the newest delivered one every time round the loop, or the
  one a rollback pinned, or the host placed by hand when nothing is delivered;
- the running host stands aside between reconciles, never inside one, by exiting
  cleanly — and returns nil so the launcher does not count it as a crash;
- a completed reconcile retires what is older than the predecessor, keeping the
  predecessor because that is what a rollback starts, and never the running one;
- rollback pins the predecessor instead of reinstalling a package: no package
  manager, no cache anyone may clean, same script on every operating system;
- the report says which host version produced it, so 'behind' is answerable.

Newest is when it arrived, never how the name sorts: '1.10' orders before '1.9',
and ordering by name would start an older host and call it an upgrade.

novox/hq ADR 0141. The delivery half — a module carrying the next host — follows;
until then nothing delivers a version and every machine takes the fallback, which
is what it does today.
2026-09-29 00:29:36 +02:00

257 lines
12 KiB
Bash
Executable File

#!/bin/sh
# Tests for nox-mesh-host-launch.
#
# The counter is the whole mechanism and it is the part to get wrong: never cleared and a node
# rolls back on a healthy boot; cleared too eagerly and it never rolls back at all. So the
# counter is what most of these assert.
set -eu
cd "$(dirname "$0")"
LAUNCH="$PWD/nox-mesh-host-launch"
PASS=0; FAIL=0
setup() {
WORK="$(mktemp -d)"
export MESH_HOST_STATE_DIR="$WORK/state"
export MESH_HOST_LIBEXEC="$WORK/libexec"
export MESH_HOST_BIN="$WORK/bin/nox-mesh-host"
export MESH_HOST_START_LIMIT=3
export MESH_HOST_BACKOFF=0
export MESH_HOST_RUN_ONCE=1
mkdir -p "$MESH_HOST_STATE_DIR" "$MESH_HOST_LIBEXEC" "$WORK/bin"
# A host that records being started. It exits immediately, which is what the launcher's
# exec makes indistinguishable from a host that ran for a week — the launcher is gone by
# then either way.
cat > "$MESH_HOST_BIN" <<'STUB'
#!/bin/sh
echo "$@" >> "$MESH_HOST_STATE_DIR/host.starts"
exit "${STUB_HOST_EXIT:-1}"
STUB
cat > "$MESH_HOST_LIBEXEC/rollback" <<'STUB'
#!/bin/sh
echo rolled-back >> "$MESH_HOST_STATE_DIR/rollback.calls"
[ -n "${STUB_ROLLBACK_FAILS:-}" ] && exit 1
echo "$(cat "$MESH_HOST_STATE_DIR/known-good" 2>/dev/null)" > "$MESH_HOST_STATE_DIR/rollback-attempted"
exit 0
STUB
chmod +x "$MESH_HOST_BIN" "$MESH_HOST_LIBEXEC/rollback"
unset STUB_ROLLBACK_FAILS || true
}
# `|| true` on every launcher call above: a launcher that exits non-zero is something to
# ASSERT, not something to abort on. With `set -e` and a bare call, removing a guard from the
# launcher killed this script at the first corrupt-counter case and silently skipped the rest —
# reporting a full pass over tests that never ran.
check() { if [ "$3" = "$4" ]; then PASS=$((PASS+1)); printf ' ok %s\n' "$1"
else FAIL=$((FAIL+1)); printf ' FAIL %s\n %s\n got: %s\n expected: %s\n' "$1" "$2" "$3" "$4"; fi; }
count() { cat "$MESH_HOST_STATE_DIR/start-attempts" 2>/dev/null || echo MISSING; }
started() { [ -f "$MESH_HOST_STATE_DIR/host.starts" ] && echo yes || echo no; }
rolled() { [ -f "$MESH_HOST_STATE_DIR/rollback.calls" ] && echo yes || echo no; }
# --- the ordinary start -----------------------------------------------------------------------
setup
"$LAUNCH" >/dev/null 2>&1 || true
check "starts the host" "the common case, every boot" "$(started)" "yes"
check "counts the attempt" "the counter is what decides a rollback later" "$(count)" "1"
check "does not roll back" "a first start is not a failure" "$(rolled)" "no"
# --- failures below the limit -----------------------------------------------------------------
setup
i=1; while [ $i -le 3 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
check "three starts do not trigger a rollback" "the limit is exceeded, not reached" "$(rolled)" "no"
check "counts them all" "" "$(count)" "3"
# --- past the limit ---------------------------------------------------------------------------
setup
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
i=1; while [ $i -le 4 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
check "the fourth start rolls back" "three failures is a binary that does not work" "$(rolled)" "yes"
check "and still starts the host" "the rolled-back version has to be run" "$(started)" "yes"
# Below the limit, not exactly zero. The rollback resets it and the rolled-back version then
# fails once here, so 1 is right — the property is that it did NOT inherit a count already at
# the limit, which would halt the new version on its first attempt.
check "resets the counter after rolling back" "the new version deserves its own attempts, or it halts at once" \
"$([ "$(count)" -lt 3 ] && echo below-limit || echo "at-limit($(count))")" "below-limit"
# --- the host clears the counter on success ----------------------------------------------------
setup
i=1; while [ $i -le 2 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
printf '0\n' > "$MESH_HOST_STATE_DIR/start-attempts" # what the host does on a completed reconcile
i=1; while [ $i -le 3 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
check "a cleared counter prevents a rollback" "a node up for months must not roll back on a healthy boot" \
"$(rolled)" "no"
# --- rolled back once already -------------------------------------------------------------------
setup
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
echo "1.4.2" > "$MESH_HOST_STATE_DIR/rollback-attempted"
i=1; while [ $i -le 4 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
check "does not roll back twice" "the previous version failing too means the machine, not the binary" \
"$(rolled)" "no"
check "halts instead" "" "$([ -f "$MESH_HOST_STATE_DIR/halted" ] && echo halted || echo running)" "halted"
# --- halted stays halted --------------------------------------------------------------------------
setup
echo "rolled back and still failing" > "$MESH_HOST_STATE_DIR/halted"
"$LAUNCH" >/dev/null 2>&1 || true
check "a halted node does not start the host" "nothing further is tried automatically" "$(started)" "no"
set +e; "$LAUNCH" >/dev/null 2>&1; RC=$?; set -e
check "a halted node exits zero" "a supervisor loop that is slow and visible beats a crash loop" "$RC" "0"
# --- the rollback itself fails ----------------------------------------------------------------------
setup
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
STUB_ROLLBACK_FAILS=1; export STUB_ROLLBACK_FAILS
i=1; while [ $i -le 4 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
check "a failed rollback halts" "restarting into the same failure would loop forever" \
"$([ -f "$MESH_HOST_STATE_DIR/halted" ] && echo halted || echo running)" "halted"
# Counted, not "was it ever started": the first three attempts DID start it, correctly, and
# only the fourth must not. An earlier version of this asserted the host was never started and
# failed for that reason rather than for a fault.
check "and does not start it on the halting attempt" "three starts, not four" \
"$(wc -l < "$MESH_HOST_STATE_DIR/host.starts" 2>/dev/null || echo 0)" "3"
# --- a corrupt counter ------------------------------------------------------------------------------
#
# The values here are chosen because they DISCRIMINATE. An earlier version used
# "not-a-number", which shell arithmetic happens to evaluate to 0 — so the test passed with the
# guard removed and proved nothing. These two do not:
#
# 5x shell arithmetic errors, and under `set -e` the launcher dies without starting the host
# 0x10 is read as HEX 16 — past the limit, so a healthy node would roll back for no reason
for corrupt in "5x" "0x10" "1 2" ""; do
setup
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
printf '%s\n' "$corrupt" > "$MESH_HOST_STATE_DIR/start-attempts"
"$LAUNCH" >/dev/null 2>&1 || true
# The exact number is not the property — "1 2" legitimately recovers a leading 1, while
# "5x" is rejected to 0. What must hold for every one of them is that the launcher
# survives its own state and does not read it as "past the limit".
check "corrupt counter [$corrupt]: starts the host" "the launcher must not die on its own state" \
"$(started)" "yes"
check "corrupt counter [$corrupt]: does not roll back" "a healthy node must not roll back on a bad counter" \
"$(rolled)" "no"
check "corrupt counter [$corrupt]: counter is a sane integer" "it is written back for the next start to read" \
"$(count | grep -cE '^[0-9]+$')" "1"
done
# --- the loop, and shutting down ------------------------------------------------------------
#
# These need the launcher to actually run as a supervisor rather than one iteration, so they do
# not set MESH_HOST_RUN_ONCE.
# A host that exits 0 has upgraded itself and stood aside (novox/hq ADR 0005). The launcher must
# start it again — and must NOT count it, because it did not fail.
setup
unset MESH_HOST_RUN_ONCE
cat > "$MESH_HOST_BIN" <<'STUB'
#!/bin/sh
echo start >> "$MESH_HOST_STATE_DIR/host.starts"
# Exit 0 three times, then hang so the launcher stops looping and can be killed.
if [ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts")" -lt 3 ]; then exit 0; fi
sleep 30
STUB
chmod +x "$MESH_HOST_BIN"
"$LAUNCH" >/dev/null 2>&1 &
LP=$!
sleep 1
check "a clean exit restarts the host" "that is how it stands aside for a new binary" \
"$([ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts" 2>/dev/null || echo 0)" -ge 3 ] && echo looped || echo stopped)" "looped"
# No counter file at all: nothing has failed, so nothing has been counted.
check "a clean exit is not counted as a failure" "it finished, it did not fail" "$(count)" "MISSING"
# Shutting down: the signal must reach the host, and the launcher must wait for it rather than
# exiting and leaving the host to be killed mid-apply.
kill -TERM "$LP" 2>/dev/null
sleep 1
check "SIGTERM stops the launcher" "a supervisor that ignores shutdown hangs the machine" \
"$(kill -0 "$LP" 2>/dev/null && echo running || echo stopped)" "stopped"
check "and does not leave the host running" "the child must go down with it" \
"$(pgrep -f "$MESH_HOST_BIN" >/dev/null 2>&1 && echo orphaned || echo reaped)" "reaped"
# A crash IS counted, and the launcher keeps going.
setup
unset MESH_HOST_RUN_ONCE
export MESH_HOST_BACKOFF=0
cat > "$MESH_HOST_BIN" <<'STUB'
#!/bin/sh
echo start >> "$MESH_HOST_STATE_DIR/host.starts"
if [ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts")" -lt 2 ]; then exit 3; fi
sleep 30
STUB
chmod +x "$MESH_HOST_BIN"
"$LAUNCH" >/dev/null 2>&1 &
LP=$!
sleep 1
check "a crash is counted" "unlike a clean exit, which is not" "$(count)" "1"
kill -TERM "$LP" 2>/dev/null; sleep 1; pkill -f "$MESH_HOST_BIN" 2>/dev/null || true
# --- which version it runs (novox/hq ADR 0141) ------------------------------------------------
#
# Versions live side by side in directories named for them. The launcher picks one every time round
# the loop, never once: standing aside for a successor is a clean exit, and the next turn has to run
# what is on disk NOW — resolved once, the same binary would restart for ever and no upgrade would
# ever take.
# deliver a version as the mesh would, recording which one ran so a test can assert the choice.
deliver() {
mkdir -p "$MESH_HOST_LIBEXEC/versions/$1"
cat > "$MESH_HOST_LIBEXEC/versions/$1/nox-mesh-host" <<STUB
#!/bin/sh
echo "$1" >> "\$MESH_HOST_STATE_DIR/which.ran"
exit "\${STUB_HOST_EXIT:-1}"
STUB
chmod +x "$MESH_HOST_LIBEXEC/versions/$1/nox-mesh-host"
# When it arrived is what "newest" means, so it is set rather than left to the clock.
touch -d "$2" "$MESH_HOST_LIBEXEC/versions/$1/nox-mesh-host" "$MESH_HOST_LIBEXEC/versions/$1"
}
which_ran() { cat "$MESH_HOST_STATE_DIR/which.ran" 2>/dev/null || echo NONE; }
# Newest is when it arrived, not how its name sorts: "1.10" orders before "1.9" by name, so ordering
# by name would run an older host and call it an upgrade.
setup
deliver 1.10 "2 hours ago"
deliver 1.9 "1 hour ago"
"$LAUNCH" >/dev/null 2>&1 || true
check "runs the newest delivered version" "newest is when it arrived, not how the name sorts" \
"$(which_ran)" "1.9"
# A pin from a rollback beats the newest, or the launcher would start the failing binary again and
# the rollback would flap.
setup
deliver 1.9 "2 hours ago"
deliver 2.0 "1 hour ago"
echo 1.9 > "$MESH_HOST_STATE_DIR/rollback-pinned"
"$LAUNCH" >/dev/null 2>&1 || true
check "a pinned version beats the newest" "otherwise a rollback starts the binary it just rejected" \
"$(which_ran)" "1.9"
# A pin naming a version that is not there is ignored rather than fatal: the machine choosing for
# itself is better than a machine that starts nothing.
setup
deliver 2.0 "1 hour ago"
echo 1.9 > "$MESH_HOST_STATE_DIR/rollback-pinned"
"$LAUNCH" >/dev/null 2>&1 || true
check "an undeliverable pin is ignored" "a machine that starts nothing is worse than one that chooses" \
"$(which_ran)" "2.0"
# An interrupted delivery leaves a directory with no binary in it. Treating it as the newest would
# mean running nothing.
setup
deliver 1.9 "2 hours ago"
mkdir -p "$MESH_HOST_LIBEXEC/versions/2.0-half"
touch -d "1 minute ago" "$MESH_HOST_LIBEXEC/versions/2.0-half"
"$LAUNCH" >/dev/null 2>&1 || true
check "skips a version with no binary" "a directory is not a version; the binary is" \
"$(which_ran)" "1.9"
# Nothing delivered: the host placed by hand, which is how the first one always arrives. Without this
# the change would strand every machine in the mesh on the day it ships.
setup
"$LAUNCH" >/dev/null 2>&1 || true
check "falls back to the host placed by hand" "every first host arrives this way" "$(started)" "yes"
printf '\nlaunch: %d passed, %d failed\n' "$PASS" "$FAIL"
[ "$FAIL" -eq 0 ]