#!/bin/sh
# Supervise the host: start it, watch it, and decide what to do when it stops.
#
# novox/hq ADR 0005. The init is asked for ONE thing — run this at boot — and everything else
# lives here, in a script that can be tested. Whether to restart, how long to wait, when to give
# up, when to roll back: all of it is policy, and policy in a unit file can only be read and
# hoped for.
#
# It does NOT exec the host. Exec would replace this process, and then only the init could
# restart anything — which is the arrangement this exists to remove. The cost of staying is
# signal handling, below.
#
# POSIX sh. `set -e` is deliberately absent: this script's whole job is to inspect exit codes,
# and -e would make it exit on the first one it is meant to handle.
set -u

STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}"
LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}"
HOST="${MESH_HOST_BIN:-/usr/bin/nox-mesh-host}"
LIMIT="${MESH_HOST_START_LIMIT:-3}"
BACKOFF="${MESH_HOST_BACKOFF:-5}"
ONCE="${MESH_HOST_RUN_ONCE:-}"   # tests run one iteration; nothing else sets this

ATTEMPTS="$STATE_DIR/start-attempts"
HALTED="$STATE_DIR/halted"

say() { echo "nox-mesh-host-launch: $*" >&2; }

child=
stopping=

# The machine is shutting down. Pass it on and wait for the host to finish — a supervisor that
# exits while its child is still running leaves the host to be killed rather than to stop, and
# an apply interrupted that way is exactly the half-configured machine this project is about.
on_term() {
	stopping=yes
	if [ -n "$child" ]; then
		say "stopping: passing the signal to the host"
		kill -TERM "$child" 2>/dev/null
	fi
}
trap on_term TERM INT

mkdir -p "$STATE_DIR"

while :; do
	if [ -n "$stopping" ]; then
		exit 0
	fi

	if [ -e "$HALTED" ]; then
		say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')"
		say "not starting the host. this node needs a person."
		exit 0
	fi

	# Consecutive failed starts, not starts. Cleared by the host itself when it completes a
	# reconcile, which is the only evidence either this or known-good has.
	#
	# Read the FIRST FIELD, then insist it is a plain integer.
	#
	# Stripping whitespace instead concatenates, and that is not hypothetical: a counter
	# holding "1 2" became "12", past the limit, so a healthy node rolled itself back. An
	# unreadable counter must fail towards "start normally", never towards "give up".
	count=0
	if [ -s "$ATTEMPTS" ]; then
		read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0
	fi
	case "${count:-}" in
		'' | *[!0-9]*) count=0 ;;
	esac

	if [ "$count" -ge "$LIMIT" ]; then
		if [ -e "$STATE_DIR/rollback-attempted" ]; then
			say "the host failed $count times after a rollback. the previous version does not"
			say "start either, so this is the machine and not the binary."
			printf 'rolled back and still failing\n' > "$HALTED"
			exit 0
		fi

		say "the host failed $count times. rolling back."
		if "$LIBEXEC/rollback"; then
			# Fresh count for the version just installed: it deserves its own attempts, and
			# without this it inherits a count already over the limit and halts at once.
			#
			# The variable too, not only the file. Resetting one and not the other made the
			# next failure count from the OLD value — so the rolled-back version got one
			# attempt instead of three.
			count=0
			printf '%s\n' "$count" > "$ATTEMPTS"
		else
			say "rollback failed. halting rather than restarting into the same failure."
			printf 'rollback failed\n' > "$HALTED"
			exit 0
		fi
	fi

	"$HOST" run &
	child=$!
	status=0
	wait "$child" || status=$?
	child=

	if [ -n "$stopping" ]; then
		exit 0
	fi

	# A signal the host did not survive, and we are not shutting down: treat it as a crash.
	case "$status" in
		0)
			# Exited cleanly. That is how the host stands aside for a new binary after an
			# upgrade (novox/hq ADR 0005) — so loop and run whatever is now on disk.
			#
			# Deliberately NOT counted, and this is the whole reason the counter is
			# incremented here rather than before the start: counting attempts meant a host
			# that upgraded itself three times rolled itself back, having worked perfectly
			# every time.
			say "the host exited cleanly; starting it again"
			continue
			;;
	esac

	count=$((count + 1))
	printf '%s\n' "$count" > "$ATTEMPTS"

	say "the host exited $status ($count consecutive); restarting in ${BACKOFF}s"
	[ -n "$ONCE" ] && exit "$status"
	sleep "$BACKOFF"
done
