The supervision was already right: a clean exit means the host stood aside, and the launcher's next turn runs what is on disk. Two things made it dead code — nothing told the running host a successor was waiting, and the rollback resolved its known-good version through pacman, which no machine here uses and which two of three operating systems do not have. Keeping a version rather than a path was the clue. Versions now live in directories named for them: - the launcher picks the newest delivered one every time round the loop, or the one a rollback pinned, or the host placed by hand when nothing is delivered; - the running host stands aside between reconciles, never inside one, by exiting cleanly — and returns nil so the launcher does not count it as a crash; - a completed reconcile retires what is older than the predecessor, keeping the predecessor because that is what a rollback starts, and never the running one; - rollback pins the predecessor instead of reinstalling a package: no package manager, no cache anyone may clean, same script on every operating system; - the report says which host version produced it, so 'behind' is answerable. Newest is when it arrived, never how the name sorts: '1.10' orders before '1.9', and ordering by name would start an older host and call it an upgrade. novox/hq ADR 0141. The delivery half — a module carrying the next host — follows; until then nothing delivers a version and every machine takes the fallback, which is what it does today.
257 lines
12 KiB
Bash
Executable File
257 lines
12 KiB
Bash
Executable File
#!/bin/sh
|
|
# Tests for nox-mesh-host-launch.
|
|
#
|
|
# The counter is the whole mechanism and it is the part to get wrong: never cleared and a node
|
|
# rolls back on a healthy boot; cleared too eagerly and it never rolls back at all. So the
|
|
# counter is what most of these assert.
|
|
set -eu
|
|
cd "$(dirname "$0")"
|
|
LAUNCH="$PWD/nox-mesh-host-launch"
|
|
PASS=0; FAIL=0
|
|
|
|
setup() {
|
|
WORK="$(mktemp -d)"
|
|
export MESH_HOST_STATE_DIR="$WORK/state"
|
|
export MESH_HOST_LIBEXEC="$WORK/libexec"
|
|
export MESH_HOST_BIN="$WORK/bin/nox-mesh-host"
|
|
export MESH_HOST_START_LIMIT=3
|
|
export MESH_HOST_BACKOFF=0
|
|
export MESH_HOST_RUN_ONCE=1
|
|
mkdir -p "$MESH_HOST_STATE_DIR" "$MESH_HOST_LIBEXEC" "$WORK/bin"
|
|
|
|
# A host that records being started. It exits immediately, which is what the launcher's
|
|
# exec makes indistinguishable from a host that ran for a week — the launcher is gone by
|
|
# then either way.
|
|
cat > "$MESH_HOST_BIN" <<'STUB'
|
|
#!/bin/sh
|
|
echo "$@" >> "$MESH_HOST_STATE_DIR/host.starts"
|
|
exit "${STUB_HOST_EXIT:-1}"
|
|
STUB
|
|
cat > "$MESH_HOST_LIBEXEC/rollback" <<'STUB'
|
|
#!/bin/sh
|
|
echo rolled-back >> "$MESH_HOST_STATE_DIR/rollback.calls"
|
|
[ -n "${STUB_ROLLBACK_FAILS:-}" ] && exit 1
|
|
echo "$(cat "$MESH_HOST_STATE_DIR/known-good" 2>/dev/null)" > "$MESH_HOST_STATE_DIR/rollback-attempted"
|
|
exit 0
|
|
STUB
|
|
chmod +x "$MESH_HOST_BIN" "$MESH_HOST_LIBEXEC/rollback"
|
|
unset STUB_ROLLBACK_FAILS || true
|
|
}
|
|
|
|
# `|| true` on every launcher call above: a launcher that exits non-zero is something to
|
|
# ASSERT, not something to abort on. With `set -e` and a bare call, removing a guard from the
|
|
# launcher killed this script at the first corrupt-counter case and silently skipped the rest —
|
|
# reporting a full pass over tests that never ran.
|
|
check() { if [ "$3" = "$4" ]; then PASS=$((PASS+1)); printf ' ok %s\n' "$1"
|
|
else FAIL=$((FAIL+1)); printf ' FAIL %s\n %s\n got: %s\n expected: %s\n' "$1" "$2" "$3" "$4"; fi; }
|
|
|
|
count() { cat "$MESH_HOST_STATE_DIR/start-attempts" 2>/dev/null || echo MISSING; }
|
|
started() { [ -f "$MESH_HOST_STATE_DIR/host.starts" ] && echo yes || echo no; }
|
|
rolled() { [ -f "$MESH_HOST_STATE_DIR/rollback.calls" ] && echo yes || echo no; }
|
|
|
|
# --- the ordinary start -----------------------------------------------------------------------
|
|
setup
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
check "starts the host" "the common case, every boot" "$(started)" "yes"
|
|
check "counts the attempt" "the counter is what decides a rollback later" "$(count)" "1"
|
|
check "does not roll back" "a first start is not a failure" "$(rolled)" "no"
|
|
|
|
# --- failures below the limit -----------------------------------------------------------------
|
|
setup
|
|
i=1; while [ $i -le 3 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
|
|
check "three starts do not trigger a rollback" "the limit is exceeded, not reached" "$(rolled)" "no"
|
|
check "counts them all" "" "$(count)" "3"
|
|
|
|
# --- past the limit ---------------------------------------------------------------------------
|
|
setup
|
|
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
|
|
i=1; while [ $i -le 4 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
|
|
check "the fourth start rolls back" "three failures is a binary that does not work" "$(rolled)" "yes"
|
|
check "and still starts the host" "the rolled-back version has to be run" "$(started)" "yes"
|
|
# Below the limit, not exactly zero. The rollback resets it and the rolled-back version then
|
|
# fails once here, so 1 is right — the property is that it did NOT inherit a count already at
|
|
# the limit, which would halt the new version on its first attempt.
|
|
check "resets the counter after rolling back" "the new version deserves its own attempts, or it halts at once" \
|
|
"$([ "$(count)" -lt 3 ] && echo below-limit || echo "at-limit($(count))")" "below-limit"
|
|
|
|
# --- the host clears the counter on success ----------------------------------------------------
|
|
setup
|
|
i=1; while [ $i -le 2 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
|
|
printf '0\n' > "$MESH_HOST_STATE_DIR/start-attempts" # what the host does on a completed reconcile
|
|
i=1; while [ $i -le 3 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
|
|
check "a cleared counter prevents a rollback" "a node up for months must not roll back on a healthy boot" \
|
|
"$(rolled)" "no"
|
|
|
|
# --- rolled back once already -------------------------------------------------------------------
|
|
setup
|
|
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
|
|
echo "1.4.2" > "$MESH_HOST_STATE_DIR/rollback-attempted"
|
|
i=1; while [ $i -le 4 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
|
|
check "does not roll back twice" "the previous version failing too means the machine, not the binary" \
|
|
"$(rolled)" "no"
|
|
check "halts instead" "" "$([ -f "$MESH_HOST_STATE_DIR/halted" ] && echo halted || echo running)" "halted"
|
|
|
|
# --- halted stays halted --------------------------------------------------------------------------
|
|
setup
|
|
echo "rolled back and still failing" > "$MESH_HOST_STATE_DIR/halted"
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
check "a halted node does not start the host" "nothing further is tried automatically" "$(started)" "no"
|
|
set +e; "$LAUNCH" >/dev/null 2>&1; RC=$?; set -e
|
|
check "a halted node exits zero" "a supervisor loop that is slow and visible beats a crash loop" "$RC" "0"
|
|
|
|
# --- the rollback itself fails ----------------------------------------------------------------------
|
|
setup
|
|
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
|
|
STUB_ROLLBACK_FAILS=1; export STUB_ROLLBACK_FAILS
|
|
i=1; while [ $i -le 4 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
|
|
check "a failed rollback halts" "restarting into the same failure would loop forever" \
|
|
"$([ -f "$MESH_HOST_STATE_DIR/halted" ] && echo halted || echo running)" "halted"
|
|
# Counted, not "was it ever started": the first three attempts DID start it, correctly, and
|
|
# only the fourth must not. An earlier version of this asserted the host was never started and
|
|
# failed for that reason rather than for a fault.
|
|
check "and does not start it on the halting attempt" "three starts, not four" \
|
|
"$(wc -l < "$MESH_HOST_STATE_DIR/host.starts" 2>/dev/null || echo 0)" "3"
|
|
|
|
# --- a corrupt counter ------------------------------------------------------------------------------
|
|
#
|
|
# The values here are chosen because they DISCRIMINATE. An earlier version used
|
|
# "not-a-number", which shell arithmetic happens to evaluate to 0 — so the test passed with the
|
|
# guard removed and proved nothing. These two do not:
|
|
#
|
|
# 5x shell arithmetic errors, and under `set -e` the launcher dies without starting the host
|
|
# 0x10 is read as HEX 16 — past the limit, so a healthy node would roll back for no reason
|
|
for corrupt in "5x" "0x10" "1 2" ""; do
|
|
setup
|
|
echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
|
|
printf '%s\n' "$corrupt" > "$MESH_HOST_STATE_DIR/start-attempts"
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
# The exact number is not the property — "1 2" legitimately recovers a leading 1, while
|
|
# "5x" is rejected to 0. What must hold for every one of them is that the launcher
|
|
# survives its own state and does not read it as "past the limit".
|
|
check "corrupt counter [$corrupt]: starts the host" "the launcher must not die on its own state" \
|
|
"$(started)" "yes"
|
|
check "corrupt counter [$corrupt]: does not roll back" "a healthy node must not roll back on a bad counter" \
|
|
"$(rolled)" "no"
|
|
check "corrupt counter [$corrupt]: counter is a sane integer" "it is written back for the next start to read" \
|
|
"$(count | grep -cE '^[0-9]+$')" "1"
|
|
done
|
|
|
|
# --- the loop, and shutting down ------------------------------------------------------------
|
|
#
|
|
# These need the launcher to actually run as a supervisor rather than one iteration, so they do
|
|
# not set MESH_HOST_RUN_ONCE.
|
|
|
|
# A host that exits 0 has upgraded itself and stood aside (novox/hq ADR 0005). The launcher must
|
|
# start it again — and must NOT count it, because it did not fail.
|
|
setup
|
|
unset MESH_HOST_RUN_ONCE
|
|
cat > "$MESH_HOST_BIN" <<'STUB'
|
|
#!/bin/sh
|
|
echo start >> "$MESH_HOST_STATE_DIR/host.starts"
|
|
# Exit 0 three times, then hang so the launcher stops looping and can be killed.
|
|
if [ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts")" -lt 3 ]; then exit 0; fi
|
|
sleep 30
|
|
STUB
|
|
chmod +x "$MESH_HOST_BIN"
|
|
"$LAUNCH" >/dev/null 2>&1 &
|
|
LP=$!
|
|
sleep 1
|
|
check "a clean exit restarts the host" "that is how it stands aside for a new binary" \
|
|
"$([ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts" 2>/dev/null || echo 0)" -ge 3 ] && echo looped || echo stopped)" "looped"
|
|
# No counter file at all: nothing has failed, so nothing has been counted.
|
|
check "a clean exit is not counted as a failure" "it finished, it did not fail" "$(count)" "MISSING"
|
|
|
|
# Shutting down: the signal must reach the host, and the launcher must wait for it rather than
|
|
# exiting and leaving the host to be killed mid-apply.
|
|
kill -TERM "$LP" 2>/dev/null
|
|
sleep 1
|
|
check "SIGTERM stops the launcher" "a supervisor that ignores shutdown hangs the machine" \
|
|
"$(kill -0 "$LP" 2>/dev/null && echo running || echo stopped)" "stopped"
|
|
check "and does not leave the host running" "the child must go down with it" \
|
|
"$(pgrep -f "$MESH_HOST_BIN" >/dev/null 2>&1 && echo orphaned || echo reaped)" "reaped"
|
|
|
|
# A crash IS counted, and the launcher keeps going.
|
|
setup
|
|
unset MESH_HOST_RUN_ONCE
|
|
export MESH_HOST_BACKOFF=0
|
|
cat > "$MESH_HOST_BIN" <<'STUB'
|
|
#!/bin/sh
|
|
echo start >> "$MESH_HOST_STATE_DIR/host.starts"
|
|
if [ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts")" -lt 2 ]; then exit 3; fi
|
|
sleep 30
|
|
STUB
|
|
chmod +x "$MESH_HOST_BIN"
|
|
"$LAUNCH" >/dev/null 2>&1 &
|
|
LP=$!
|
|
sleep 1
|
|
check "a crash is counted" "unlike a clean exit, which is not" "$(count)" "1"
|
|
kill -TERM "$LP" 2>/dev/null; sleep 1; pkill -f "$MESH_HOST_BIN" 2>/dev/null || true
|
|
|
|
# --- which version it runs (novox/hq ADR 0141) ------------------------------------------------
|
|
#
|
|
# Versions live side by side in directories named for them. The launcher picks one every time round
|
|
# the loop, never once: standing aside for a successor is a clean exit, and the next turn has to run
|
|
# what is on disk NOW — resolved once, the same binary would restart for ever and no upgrade would
|
|
# ever take.
|
|
|
|
# deliver a version as the mesh would, recording which one ran so a test can assert the choice.
|
|
deliver() {
|
|
mkdir -p "$MESH_HOST_LIBEXEC/versions/$1"
|
|
cat > "$MESH_HOST_LIBEXEC/versions/$1/nox-mesh-host" <<STUB
|
|
#!/bin/sh
|
|
echo "$1" >> "\$MESH_HOST_STATE_DIR/which.ran"
|
|
exit "\${STUB_HOST_EXIT:-1}"
|
|
STUB
|
|
chmod +x "$MESH_HOST_LIBEXEC/versions/$1/nox-mesh-host"
|
|
# When it arrived is what "newest" means, so it is set rather than left to the clock.
|
|
touch -d "$2" "$MESH_HOST_LIBEXEC/versions/$1/nox-mesh-host" "$MESH_HOST_LIBEXEC/versions/$1"
|
|
}
|
|
which_ran() { cat "$MESH_HOST_STATE_DIR/which.ran" 2>/dev/null || echo NONE; }
|
|
|
|
# Newest is when it arrived, not how its name sorts: "1.10" orders before "1.9" by name, so ordering
|
|
# by name would run an older host and call it an upgrade.
|
|
setup
|
|
deliver 1.10 "2 hours ago"
|
|
deliver 1.9 "1 hour ago"
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
check "runs the newest delivered version" "newest is when it arrived, not how the name sorts" \
|
|
"$(which_ran)" "1.9"
|
|
|
|
# A pin from a rollback beats the newest, or the launcher would start the failing binary again and
|
|
# the rollback would flap.
|
|
setup
|
|
deliver 1.9 "2 hours ago"
|
|
deliver 2.0 "1 hour ago"
|
|
echo 1.9 > "$MESH_HOST_STATE_DIR/rollback-pinned"
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
check "a pinned version beats the newest" "otherwise a rollback starts the binary it just rejected" \
|
|
"$(which_ran)" "1.9"
|
|
|
|
# A pin naming a version that is not there is ignored rather than fatal: the machine choosing for
|
|
# itself is better than a machine that starts nothing.
|
|
setup
|
|
deliver 2.0 "1 hour ago"
|
|
echo 1.9 > "$MESH_HOST_STATE_DIR/rollback-pinned"
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
check "an undeliverable pin is ignored" "a machine that starts nothing is worse than one that chooses" \
|
|
"$(which_ran)" "2.0"
|
|
|
|
# An interrupted delivery leaves a directory with no binary in it. Treating it as the newest would
|
|
# mean running nothing.
|
|
setup
|
|
deliver 1.9 "2 hours ago"
|
|
mkdir -p "$MESH_HOST_LIBEXEC/versions/2.0-half"
|
|
touch -d "1 minute ago" "$MESH_HOST_LIBEXEC/versions/2.0-half"
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
check "skips a version with no binary" "a directory is not a version; the binary is" \
|
|
"$(which_ran)" "1.9"
|
|
|
|
# Nothing delivered: the host placed by hand, which is how the first one always arrives. Without this
|
|
# the change would strand every machine in the mesh on the day it ships.
|
|
setup
|
|
"$LAUNCH" >/dev/null 2>&1 || true
|
|
check "falls back to the host placed by hand" "every first host arrives this way" "$(started)" "yes"
|
|
|
|
printf '\nlaunch: %d passed, %d failed\n' "$PASS" "$FAIL"
|
|
[ "$FAIL" -eq 0 ]
|