#!/bin/sh
# Supervise the host: start it, watch it, and decide what to do when it stops.
#
# novox/hq ADR 0005. The init is asked for ONE thing — run this at boot — and everything else
# lives here, in a script that can be tested. Whether to restart, how long to wait, when to give
# up, when to roll back: all of it is policy, and policy in a unit file can only be read and
# hoped for.
#
# **It is the witness of the host's own successor** (novox/hq to-be 45 §8, ADR 0227 rule 8). A
# delivered host that is not the known-good one runs on trial: it must report its declaration
# under its own build — which it marks by writing itself into known-good — within a bound. One
# that crashes repeatedly, exits without standing aside for anything, or does not report in
# bound is stopped, recorded in `rolled-back` with why, and the known-good one is run. The host
# says every record on its reports, so the mesh raises it as a condition. One rollback per
# version: a version in `rolled-back` is never started again; a newer delivery is.
#
# It does NOT exec the host. Exec would replace this process, and then only the init could
# restart anything — which is the arrangement this exists to remove. The cost of staying is
# signal handling, below.
#
# Everything a rollback does is one of: read a file, look at a directory, write a file. It shares
# no code with the host and calls none of it: a binary that cannot start cannot be its own
# recovery.
#
# POSIX sh. `set -e` is deliberately absent: this script's whole job is to inspect exit codes,
# and -e would make it exit on the first one it is meant to handle.
set -u

STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}"
LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}"
# The host that was placed by hand, used only when nothing has been delivered. The first host on a
# machine always arrives this way; every one after it is delivered (novox/hq ADR 0141).
FALLBACK="${MESH_HOST_BIN:-/usr/bin/nox-mesh-host}"
VERSIONS="$LIBEXEC/versions"
BINARY="nox-mesh-host"
LIMIT="${MESH_HOST_START_LIMIT:-3}"
BACKOFF="${MESH_HOST_BACKOFF:-5}"
# How long a host on trial has to report, in seconds: to-be 45 §8's ten minutes from the apply.
BOUND="${MESH_HOST_TRIAL_BOUND:-600}"
TICK="${MESH_HOST_TRIAL_TICK:-5}"
GRACE="${MESH_HOST_STOP_GRACE:-20}"
ONCE="${MESH_HOST_RUN_ONCE:-}"   # tests run one iteration; nothing else sets this

ATTEMPTS="$STATE_DIR/start-attempts"
HALTED="$STATE_DIR/halted"
PINNED="$STATE_DIR/rollback-pinned"
KNOWN_GOOD="$STATE_DIR/known-good"
# One line per verdict: from, to, when (seconds since the epoch), outcome, why — tab-separated, read
# by the host (internal/upgrade) and said on its reports.
ROLLED="$STATE_DIR/rolled-back"
TRIAL="$STATE_DIR/trial"
EXPIRED="$STATE_DIR/trial-expired"

say() { echo "nox-mesh-host-launch: $*" >&2; }

now() { date +%s; }

known_good() { tr -d '[:space:]' < "$KNOWN_GOOD" 2>/dev/null || true; }

delivered() { [ -n "$1" ] && [ -x "$VERSIONS/$1/$BINARY" ]; }

rolled_back() { [ -s "$ROLLED" ] && cut -f1 "$ROLLED" 2>/dev/null | grep -qxF "$1"; }

# The version a host path is: the directory it was delivered in, or nothing for the host placed by hand.
version_of() {
	case "$1" in
		"$VERSIONS"/*/"$BINARY") v="${1#"$VERSIONS"/}"; echo "${v%/"$BINARY"}" ;;
		*) echo "" ;;
	esac
}

record() { # from to outcome why
	printf '%s\t%s\t%s\t%s\t%s\n' "$1" "$2" "$(now)" "$3" "$4" >> "$ROLLED"
}

# Which host to run: the newest delivered one that was never rolled back — or, while a rollback's pin
# stands, the version it pinned — or the one placed by hand when nothing has been delivered.
#
# **Asked every time round the loop, not once.** Standing aside for a successor is a clean exit, and
# the next turn has to run what is on disk NOW — resolving this once would restart the same binary
# for ever and the upgrade would never take.
#
# Newest by when it arrived, never by how its name sorts: a version string is whatever the source was
# described as, and those do not sort — "1.10" orders before "1.9". Ordering by name would start an
# older host and call it an upgrade.
#
# **A pin stands until something newer arrives.** A version delivered after the pin was written, and
# never rolled back, is a new build the mesh asks for: the pin goes and it runs, on trial.
pick_host() {
	newest=
	# A directory with no executable in it is not a version: an interrupted delivery leaves one, and
	# running "the newest" would then mean running nothing.
	for candidate in $(ls -1t "$VERSIONS" 2>/dev/null || true); do
		delivered "$candidate" || continue
		rolled_back "$candidate" && continue
		newest="$candidate"
		break
	done
	if [ -s "$PINNED" ]; then
		pinned="$(tr -d '[:space:]' < "$PINNED" 2>/dev/null || true)"
		if [ -n "$newest" ] && [ "$newest" != "$pinned" ] && [ "$VERSIONS/$newest" -nt "$PINNED" ]; then
			say "host $newest arrived after the rollback to $pinned; running it"
			rm -f "$PINNED"
		elif delivered "$pinned"; then
			echo "$VERSIONS/$pinned/$BINARY"
			return 0
		else
			say "the pinned version '$pinned' is not delivered; ignoring the pin"
		fi
	fi
	if [ -n "$newest" ]; then
		echo "$VERSIONS/$newest/$BINARY"
		return 0
	fi
	echo "$FALLBACK"
}

# Go back from `from` to the known-good version, once, and say so. False when there is nowhere to go.
roll_back() { # from why
	kg="$(known_good)"
	if [ -z "$1" ] || [ "$1" = "$kg" ] || ! delivered "$kg"; then
		return 1
	fi
	record "$1" "$kg" rolled-back "$2"
	# The pin is what stops the launcher starting a version newer than known-good that is not the one
	# rolled back; the record is what stops it starting this one again.
	printf '%s\n' "$kg" > "$PINNED"
	rm -f "$TRIAL"
	say "rolled back from host $1 to $kg: $2"
	return 0
}

# Watch a host on trial: done when it writes itself into known-good, stopped when the bound passes.
# A subshell beside the host, so the launcher's own wait is untouched.
trial_watch() { # pid version deadline
	while kill -0 "$1" 2>/dev/null; do
		[ "$(known_good)" = "$2" ] && return 0
		if [ "$(now)" -ge "$3" ]; then
			printf '%s\n' "$2" > "$EXPIRED"
			say "host $2 did not report within ${BOUND}s of starting; stopping it"
			kill -TERM "$1" 2>/dev/null
			waited=0
			while kill -0 "$1" 2>/dev/null && [ "$waited" -lt "$GRACE" ]; do
				sleep 1
				waited=$((waited + 1))
			done
			kill -KILL "$1" 2>/dev/null
			return 0
		fi
		sleep "$TICK"
	done
}

child=
watcher=
stopping=

# The machine is shutting down. Pass it on and wait for the host to finish — a supervisor that
# exits while its child is still running leaves the host to be killed rather than to stop, and
# an apply interrupted that way is exactly the half-configured machine this project is about.
on_term() {
	stopping=yes
	if [ -n "$child" ]; then
		say "stopping: passing the signal to the host"
		kill -TERM "$child" 2>/dev/null
	fi
}
trap on_term TERM INT

mkdir -p "$STATE_DIR"
# What this launcher is, so a delivered successor of it is run at the next clean exit rather than at
# the next boot.
SELF="$(cksum < "$0" 2>/dev/null || true)"

while :; do
	if [ -n "$stopping" ]; then
		exit 0
	fi

	if [ -e "$HALTED" ]; then
		say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')"
		say "not starting the host. this node needs a person."
		exit 0
	fi

	# Consecutive failed starts, not starts. Cleared by the host itself when it reports, which is
	# the only evidence either this or known-good has.
	#
	# Read the FIRST FIELD, then insist it is a plain integer.
	#
	# Stripping whitespace instead concatenates, and that is not hypothetical: a counter
	# holding "1 2" became "12", past the limit, so a healthy node rolled itself back. An
	# unreadable counter must fail towards "start normally", never towards "give up".
	count=0
	if [ -s "$ATTEMPTS" ]; then
		read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0
	fi
	case "${count:-}" in
		'' | *[!0-9]*) count=0 ;;
	esac

	HOST="$(pick_host)"
	version="$(version_of "$HOST")"

	if [ "$count" -ge "$LIMIT" ]; then
		if roll_back "$version" "it failed $count times in a row"; then
			# Fresh count for the version now run: it deserves its own attempts, and without this
			# it inherits a count already over the limit and halts at once. The variable too, not
			# only the file: resetting one and not the other made the next failure count from the
			# OLD value — so the rolled-back version got one attempt instead of three.
			count=0
			printf '%s\n' "$count" > "$ATTEMPTS"
			continue
		fi
		kg="$(known_good)"
		if [ -n "$kg" ] && { [ "$version" = "$kg" ] || [ -z "$version" ]; } && [ -s "$ROLLED" ]; then
			# Already gone back, and what it went back to fails as well: this is the machine and
			# not a binary. Rolling back again would flap between two versions for ever.
			say "the host failed $count times after a rollback. the version it went back to does"
			say "not start either, so this is the machine and not the binary."
			record "${version:-placed-by-hand}" "" halted "the version rolled back to failed $count times as well"
			printf 'rolled back and still failing\n' > "$HALTED"
			exit 0
		fi
		# Nothing to go back to: no host has ever reported here, or the one that did was placed by
		# hand and is not delivered. An installation failure rather than an upgrade's — said, and
		# tried again slowly, never guessed at.
		say "the host failed $count times and there is no delivered known-good version to go back to."
		count=0
		printf '%s\n' "$count" > "$ATTEMPTS"
	fi

	if [ ! -x "$HOST" ]; then
		say "no host to run: nothing delivered under $VERSIONS and $FALLBACK is not executable."
		printf 'no host binary\n' > "$HALTED"
		exit 0
	fi

	# **On trial**: a delivered version that is not the known-good one, while a known-good one is
	# delivered to go back to. The trial starts when this version was first started, and survives
	# this launcher restarting.
	trial=
	deadline=
	kg="$(known_good)"
	if [ -n "$version" ] && [ "$version" != "$kg" ] && delivered "$kg"; then
		trial=yes
		started=
		if [ -s "$TRIAL" ]; then
			read -r on since _ < "$TRIAL" 2>/dev/null || on=
			[ "${on:-}" = "$version" ] && started="${since:-}"
		fi
		case "$started" in
			'' | *[!0-9]*) started="$(now)"; printf '%s %s\n' "$version" "$started" > "$TRIAL" ;;
		esac
		deadline=$((started + BOUND))
		if [ "$(now)" -ge "$deadline" ]; then
			roll_back "$version" "it did not report within ${BOUND}s of starting" && continue
		fi
	else
		rm -f "$TRIAL"
	fi

	rm -f "$EXPIRED"
	say "running $HOST${trial:+ (on trial until it reports)}"
	"$HOST" run &
	child=$!
	watcher=
	if [ -n "$trial" ]; then
		trial_watch "$child" "$version" "$deadline" &
		watcher=$!
	fi
	status=0
	wait "$child" || status=$?
	if [ -n "$stopping" ]; then
		# The signal interrupted the wait, not the host: it was told, and is finishing what it was
		# doing. Waited for, so the service manager does not kill it half way through an apply.
		wait "$child" 2>/dev/null
	fi
	child=
	if [ -n "$watcher" ]; then
		kill "$watcher" 2>/dev/null
		wait "$watcher" 2>/dev/null
		watcher=
	fi

	if [ -n "$stopping" ]; then
		exit 0
	fi

	if [ -n "$trial" ] && [ -s "$EXPIRED" ] && [ "$(cat "$EXPIRED" 2>/dev/null)" = "$version" ]; then
		rm -f "$EXPIRED"
		roll_back "$version" "it did not report within ${BOUND}s of starting"
		count=0
		printf '%s\n' "$count" > "$ATTEMPTS"
		continue
	fi

	# A signal the host did not survive, and we are not shutting down: treat it as a crash.
	case "$status" in
		0)
			# Exited cleanly. That is how the host stands aside for a new binary after an
			# upgrade (novox/hq ADR 0005) — so loop and run whatever is now on disk.
			#
			# Deliberately NOT counted, and this is the whole reason the counter is
			# incremented here rather than before the start: counting attempts meant a host
			# that upgraded itself three times rolled itself back, having worked perfectly
			# every time.
			#
			# **Unless it stood aside for nothing.** A host on trial that exits cleanly while the
			# same host is still the one to run has not stood aside for a successor: it has
			# stopped, and a host that stops at once would otherwise loop here for ever unseen.
			if [ -n "$trial" ] && [ "$(pick_host)" = "$HOST" ] && [ "$(known_good)" != "$version" ]; then
				say "host $version exited cleanly on trial with nothing newer to stand aside for"
			else
				# A delivered successor of this launcher runs from here on, not from the next boot.
				if [ -n "$SELF" ] && [ "$(cksum < "$0" 2>/dev/null || true)" != "$SELF" ]; then
					say "the launcher was replaced; running the new one"
					exec "$0"
				fi
				say "the host exited cleanly; starting it again"
				continue
			fi
			;;
	esac

	count=$((count + 1))
	printf '%s\n' "$count" > "$ATTEMPTS"

	say "the host exited $status ($count consecutive); restarting in ${BACKOFF}s"
	[ -n "$ONCE" ] && exit "$status"
	sleep "$BACKOFF"
done
