Jochen: "I thought we did not want to run the host under a systemd/openrc/init loop, but instead had our own host-init program?" -- and that was right. I had moved the give-up logic out of unit files and left RESTART in them, with the launcher exec'ing the host and disappearing. So init still decided when the host came back, which is the arrangement 0061 exists to remove. The launcher now stays and supervises: starts the host as a child, waits, decides. Init is asked for one thing, run this at boot. There is an OpenRC script beside the systemd unit now, four lines each, which is the point -- a second init is transcription rather than a port. The cost of not exec'ing is signals. A supervisor that exits while its child runs leaves the host to be killed rather than to stop, and an apply interrupted that way is the half-configured machine this project is about. So SIGTERM is trapped, passed down, and waited on. Two bugs, both found by the tests rather than by review: A clean exit was counted as a failure. The host exits cleanly to stand aside for a new binary after an upgrade (0057), so a host that upgraded itself three times rolled itself back having worked perfectly every time. The counter now counts CONSECUTIVE FAILURES, incremented after the wait rather than before the start. And when rolling back I reset the counter file but not the variable, so the next failure counted from the old value -- the rolled-back version got one attempt instead of three. Also: the host now clears the counter when it completes a reconcile, at the same moment it records known-good and for the same reason. Without it the count only climbs, and a node up for months rolls itself back on its third ordinary restart -- a healthy machine undone by its own recovery. One test expectation was tightened rather than fixed: "resets the counter after rolling back" asserted exactly 0, which was true only under the old count-before-start semantics. It now asserts the property -- below the limit -- since 1 is correct after a rollback plus one failure. 32 launcher tests, all confirmed to bite.
130 lines
4.3 KiB
Bash
Executable File
130 lines
4.3 KiB
Bash
Executable File
#!/bin/sh
|
|
# Supervise the host: start it, watch it, and decide what to do when it stops.
|
|
#
|
|
# novox/hq ADR 0061. The init is asked for ONE thing — run this at boot — and everything else
|
|
# lives here, in a script that can be tested. Whether to restart, how long to wait, when to give
|
|
# up, when to roll back: all of it is policy, and policy in a unit file can only be read and
|
|
# hoped for.
|
|
#
|
|
# It does NOT exec the host. Exec would replace this process, and then only the init could
|
|
# restart anything — which is the arrangement this exists to remove. The cost of staying is
|
|
# signal handling, below.
|
|
#
|
|
# POSIX sh. `set -e` is deliberately absent: this script's whole job is to inspect exit codes,
|
|
# and -e would make it exit on the first one it is meant to handle.
|
|
set -u
|
|
|
|
STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}"
|
|
LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}"
|
|
HOST="${MESH_HOST_BIN:-/usr/bin/nox-mesh-host}"
|
|
LIMIT="${MESH_HOST_START_LIMIT:-3}"
|
|
BACKOFF="${MESH_HOST_BACKOFF:-5}"
|
|
ONCE="${MESH_HOST_RUN_ONCE:-}" # tests run one iteration; nothing else sets this
|
|
|
|
ATTEMPTS="$STATE_DIR/start-attempts"
|
|
HALTED="$STATE_DIR/halted"
|
|
|
|
say() { echo "nox-mesh-host-launch: $*" >&2; }
|
|
|
|
child=
|
|
stopping=
|
|
|
|
# The machine is shutting down. Pass it on and wait for the host to finish — a supervisor that
|
|
# exits while its child is still running leaves the host to be killed rather than to stop, and
|
|
# an apply interrupted that way is exactly the half-configured machine this project is about.
|
|
on_term() {
|
|
stopping=yes
|
|
if [ -n "$child" ]; then
|
|
say "stopping: passing the signal to the host"
|
|
kill -TERM "$child" 2>/dev/null
|
|
fi
|
|
}
|
|
trap on_term TERM INT
|
|
|
|
mkdir -p "$STATE_DIR"
|
|
|
|
while :; do
|
|
if [ -n "$stopping" ]; then
|
|
exit 0
|
|
fi
|
|
|
|
if [ -e "$HALTED" ]; then
|
|
say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')"
|
|
say "not starting the host. this node needs a person."
|
|
exit 0
|
|
fi
|
|
|
|
# Consecutive failed starts, not starts. Cleared by the host itself when it completes a
|
|
# reconcile, which is the only evidence either this or known-good has.
|
|
#
|
|
# Read the FIRST FIELD, then insist it is a plain integer.
|
|
#
|
|
# Stripping whitespace instead concatenates, and that is not hypothetical: a counter
|
|
# holding "1 2" became "12", past the limit, so a healthy node rolled itself back. An
|
|
# unreadable counter must fail towards "start normally", never towards "give up".
|
|
count=0
|
|
if [ -s "$ATTEMPTS" ]; then
|
|
read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0
|
|
fi
|
|
case "${count:-}" in
|
|
'' | *[!0-9]*) count=0 ;;
|
|
esac
|
|
|
|
if [ "$count" -ge "$LIMIT" ]; then
|
|
if [ -e "$STATE_DIR/rollback-attempted" ]; then
|
|
say "the host failed $count times after a rollback. the previous version does not"
|
|
say "start either, so this is the machine and not the binary."
|
|
printf 'rolled back and still failing\n' > "$HALTED"
|
|
exit 0
|
|
fi
|
|
|
|
say "the host failed $count times. rolling back."
|
|
if "$LIBEXEC/rollback"; then
|
|
# Fresh count for the version just installed: it deserves its own attempts, and
|
|
# without this it inherits a count already over the limit and halts at once.
|
|
#
|
|
# The variable too, not only the file. Resetting one and not the other made the
|
|
# next failure count from the OLD value — so the rolled-back version got one
|
|
# attempt instead of three.
|
|
count=0
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
else
|
|
say "rollback failed. halting rather than restarting into the same failure."
|
|
printf 'rollback failed\n' > "$HALTED"
|
|
exit 0
|
|
fi
|
|
fi
|
|
|
|
"$HOST" run &
|
|
child=$!
|
|
status=0
|
|
wait "$child" || status=$?
|
|
child=
|
|
|
|
if [ -n "$stopping" ]; then
|
|
exit 0
|
|
fi
|
|
|
|
# A signal the host did not survive, and we are not shutting down: treat it as a crash.
|
|
case "$status" in
|
|
0)
|
|
# Exited cleanly. That is how the host stands aside for a new binary after an
|
|
# upgrade (novox/hq ADR 0057) — so loop and run whatever is now on disk.
|
|
#
|
|
# Deliberately NOT counted, and this is the whole reason the counter is
|
|
# incremented here rather than before the start: counting attempts meant a host
|
|
# that upgraded itself three times rolled itself back, having worked perfectly
|
|
# every time.
|
|
say "the host exited cleanly; starting it again"
|
|
continue
|
|
;;
|
|
esac
|
|
|
|
count=$((count + 1))
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
|
|
say "the host exited $status ($count consecutive); restarting in ${BACKOFF}s"
|
|
[ -n "$ONCE" ] && exit "$status"
|
|
sleep "$BACKOFF"
|
|
done
|