96 comments across the two repos named records that no longer exist. Each now points at the consolidated record that holds its reasoning -- ADR 0034 (a test defends a decision) is 0017, the eight host records are 0005, the four lab records are 0016. Worth noting for next time: these are references from outside HQ, so renumbering there is not free. It cost 38 files here.
130 lines
4.3 KiB
Bash
Executable File
130 lines
4.3 KiB
Bash
Executable File
#!/bin/sh
|
|
# Supervise the host: start it, watch it, and decide what to do when it stops.
|
|
#
|
|
# novox/hq ADR 0005. The init is asked for ONE thing — run this at boot — and everything else
|
|
# lives here, in a script that can be tested. Whether to restart, how long to wait, when to give
|
|
# up, when to roll back: all of it is policy, and policy in a unit file can only be read and
|
|
# hoped for.
|
|
#
|
|
# It does NOT exec the host. Exec would replace this process, and then only the init could
|
|
# restart anything — which is the arrangement this exists to remove. The cost of staying is
|
|
# signal handling, below.
|
|
#
|
|
# POSIX sh. `set -e` is deliberately absent: this script's whole job is to inspect exit codes,
|
|
# and -e would make it exit on the first one it is meant to handle.
|
|
set -u
|
|
|
|
STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}"
|
|
LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}"
|
|
HOST="${MESH_HOST_BIN:-/usr/bin/nox-mesh-host}"
|
|
LIMIT="${MESH_HOST_START_LIMIT:-3}"
|
|
BACKOFF="${MESH_HOST_BACKOFF:-5}"
|
|
ONCE="${MESH_HOST_RUN_ONCE:-}" # tests run one iteration; nothing else sets this
|
|
|
|
ATTEMPTS="$STATE_DIR/start-attempts"
|
|
HALTED="$STATE_DIR/halted"
|
|
|
|
say() { echo "nox-mesh-host-launch: $*" >&2; }
|
|
|
|
child=
|
|
stopping=
|
|
|
|
# The machine is shutting down. Pass it on and wait for the host to finish — a supervisor that
|
|
# exits while its child is still running leaves the host to be killed rather than to stop, and
|
|
# an apply interrupted that way is exactly the half-configured machine this project is about.
|
|
on_term() {
|
|
stopping=yes
|
|
if [ -n "$child" ]; then
|
|
say "stopping: passing the signal to the host"
|
|
kill -TERM "$child" 2>/dev/null
|
|
fi
|
|
}
|
|
trap on_term TERM INT
|
|
|
|
mkdir -p "$STATE_DIR"
|
|
|
|
while :; do
|
|
if [ -n "$stopping" ]; then
|
|
exit 0
|
|
fi
|
|
|
|
if [ -e "$HALTED" ]; then
|
|
say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')"
|
|
say "not starting the host. this node needs a person."
|
|
exit 0
|
|
fi
|
|
|
|
# Consecutive failed starts, not starts. Cleared by the host itself when it completes a
|
|
# reconcile, which is the only evidence either this or known-good has.
|
|
#
|
|
# Read the FIRST FIELD, then insist it is a plain integer.
|
|
#
|
|
# Stripping whitespace instead concatenates, and that is not hypothetical: a counter
|
|
# holding "1 2" became "12", past the limit, so a healthy node rolled itself back. An
|
|
# unreadable counter must fail towards "start normally", never towards "give up".
|
|
count=0
|
|
if [ -s "$ATTEMPTS" ]; then
|
|
read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0
|
|
fi
|
|
case "${count:-}" in
|
|
'' | *[!0-9]*) count=0 ;;
|
|
esac
|
|
|
|
if [ "$count" -ge "$LIMIT" ]; then
|
|
if [ -e "$STATE_DIR/rollback-attempted" ]; then
|
|
say "the host failed $count times after a rollback. the previous version does not"
|
|
say "start either, so this is the machine and not the binary."
|
|
printf 'rolled back and still failing\n' > "$HALTED"
|
|
exit 0
|
|
fi
|
|
|
|
say "the host failed $count times. rolling back."
|
|
if "$LIBEXEC/rollback"; then
|
|
# Fresh count for the version just installed: it deserves its own attempts, and
|
|
# without this it inherits a count already over the limit and halts at once.
|
|
#
|
|
# The variable too, not only the file. Resetting one and not the other made the
|
|
# next failure count from the OLD value — so the rolled-back version got one
|
|
# attempt instead of three.
|
|
count=0
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
else
|
|
say "rollback failed. halting rather than restarting into the same failure."
|
|
printf 'rollback failed\n' > "$HALTED"
|
|
exit 0
|
|
fi
|
|
fi
|
|
|
|
"$HOST" run &
|
|
child=$!
|
|
status=0
|
|
wait "$child" || status=$?
|
|
child=
|
|
|
|
if [ -n "$stopping" ]; then
|
|
exit 0
|
|
fi
|
|
|
|
# A signal the host did not survive, and we are not shutting down: treat it as a crash.
|
|
case "$status" in
|
|
0)
|
|
# Exited cleanly. That is how the host stands aside for a new binary after an
|
|
# upgrade (novox/hq ADR 0005) — so loop and run whatever is now on disk.
|
|
#
|
|
# Deliberately NOT counted, and this is the whole reason the counter is
|
|
# incremented here rather than before the start: counting attempts meant a host
|
|
# that upgraded itself three times rolled itself back, having worked perfectly
|
|
# every time.
|
|
say "the host exited cleanly; starting it again"
|
|
continue
|
|
;;
|
|
esac
|
|
|
|
count=$((count + 1))
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
|
|
say "the host exited $status ($count consecutive); restarting in ${BACKOFF}s"
|
|
[ -n "$ONCE" ] && exit "$status"
|
|
sleep "$BACKOFF"
|
|
done
|