The launcher trusted a counter only a by-hand reconcile ever cleared and a known-good nothing in the daemon wrote, so no machine could roll its host back; the controller and the node tools were replaced in place with nothing kept. - The launcher runs a delivered host that is not known-good on trial: one that crashes, stops for nothing, or does not report within ten minutes goes back to known-good, once per version, recorded in rolled-back. The host proves itself when the mesh takes a report under its own build, says every standing verdict on its reports, never stands aside for a rolled-back version, and restarts its service once when its launcher was replaced on disk. - The engine keeps the controller's and the node tools' previous build beside the new one and judges the new one: the lease taken by the controller it started (read-only direct get of mesh-controller_lease/holder), or this machine's runtime answering $SRV.PING.node-tools.<node>, within sixty seconds of time it could ask. Not healthy: the previous restored, once, said. Proved: the previous deleted. A build declared not-reversible is never rolled back. - Retire never removes a version newer than the running one.
333 lines
13 KiB
Bash
Executable File
333 lines
13 KiB
Bash
Executable File
#!/bin/sh
|
|
# Supervise the host: start it, watch it, and decide what to do when it stops.
|
|
#
|
|
# novox/hq ADR 0005. The init is asked for ONE thing — run this at boot — and everything else
|
|
# lives here, in a script that can be tested. Whether to restart, how long to wait, when to give
|
|
# up, when to roll back: all of it is policy, and policy in a unit file can only be read and
|
|
# hoped for.
|
|
#
|
|
# **It is the witness of the host's own successor** (novox/hq to-be 45 §8, ADR 0227 rule 8). A
|
|
# delivered host that is not the known-good one runs on trial: it must report its declaration
|
|
# under its own build — which it marks by writing itself into known-good — within a bound. One
|
|
# that crashes repeatedly, exits without standing aside for anything, or does not report in
|
|
# bound is stopped, recorded in `rolled-back` with why, and the known-good one is run. The host
|
|
# says every record on its reports, so the mesh raises it as a condition. One rollback per
|
|
# version: a version in `rolled-back` is never started again; a newer delivery is.
|
|
#
|
|
# It does NOT exec the host. Exec would replace this process, and then only the init could
|
|
# restart anything — which is the arrangement this exists to remove. The cost of staying is
|
|
# signal handling, below.
|
|
#
|
|
# Everything a rollback does is one of: read a file, look at a directory, write a file. It shares
|
|
# no code with the host and calls none of it: a binary that cannot start cannot be its own
|
|
# recovery.
|
|
#
|
|
# POSIX sh. `set -e` is deliberately absent: this script's whole job is to inspect exit codes,
|
|
# and -e would make it exit on the first one it is meant to handle.
|
|
set -u
|
|
|
|
STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}"
|
|
LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}"
|
|
# The host that was placed by hand, used only when nothing has been delivered. The first host on a
|
|
# machine always arrives this way; every one after it is delivered (novox/hq ADR 0141).
|
|
FALLBACK="${MESH_HOST_BIN:-/usr/bin/nox-mesh-host}"
|
|
VERSIONS="$LIBEXEC/versions"
|
|
BINARY="nox-mesh-host"
|
|
LIMIT="${MESH_HOST_START_LIMIT:-3}"
|
|
BACKOFF="${MESH_HOST_BACKOFF:-5}"
|
|
# How long a host on trial has to report, in seconds: to-be 45 §8's ten minutes from the apply.
|
|
BOUND="${MESH_HOST_TRIAL_BOUND:-600}"
|
|
TICK="${MESH_HOST_TRIAL_TICK:-5}"
|
|
GRACE="${MESH_HOST_STOP_GRACE:-20}"
|
|
ONCE="${MESH_HOST_RUN_ONCE:-}" # tests run one iteration; nothing else sets this
|
|
|
|
ATTEMPTS="$STATE_DIR/start-attempts"
|
|
HALTED="$STATE_DIR/halted"
|
|
PINNED="$STATE_DIR/rollback-pinned"
|
|
KNOWN_GOOD="$STATE_DIR/known-good"
|
|
# One line per verdict: from, to, when (seconds since the epoch), outcome, why — tab-separated, read
|
|
# by the host (internal/upgrade) and said on its reports.
|
|
ROLLED="$STATE_DIR/rolled-back"
|
|
TRIAL="$STATE_DIR/trial"
|
|
EXPIRED="$STATE_DIR/trial-expired"
|
|
|
|
say() { echo "nox-mesh-host-launch: $*" >&2; }
|
|
|
|
now() { date +%s; }
|
|
|
|
known_good() { tr -d '[:space:]' < "$KNOWN_GOOD" 2>/dev/null || true; }
|
|
|
|
delivered() { [ -n "$1" ] && [ -x "$VERSIONS/$1/$BINARY" ]; }
|
|
|
|
rolled_back() { [ -s "$ROLLED" ] && cut -f1 "$ROLLED" 2>/dev/null | grep -qxF "$1"; }
|
|
|
|
# The version a host path is: the directory it was delivered in, or nothing for the host placed by hand.
|
|
version_of() {
|
|
case "$1" in
|
|
"$VERSIONS"/*/"$BINARY") v="${1#"$VERSIONS"/}"; echo "${v%/"$BINARY"}" ;;
|
|
*) echo "" ;;
|
|
esac
|
|
}
|
|
|
|
record() { # from to outcome why
|
|
printf '%s\t%s\t%s\t%s\t%s\n' "$1" "$2" "$(now)" "$3" "$4" >> "$ROLLED"
|
|
}
|
|
|
|
# Which host to run: the newest delivered one that was never rolled back — or, while a rollback's pin
|
|
# stands, the version it pinned — or the one placed by hand when nothing has been delivered.
|
|
#
|
|
# **Asked every time round the loop, not once.** Standing aside for a successor is a clean exit, and
|
|
# the next turn has to run what is on disk NOW — resolving this once would restart the same binary
|
|
# for ever and the upgrade would never take.
|
|
#
|
|
# Newest by when it arrived, never by how its name sorts: a version string is whatever the source was
|
|
# described as, and those do not sort — "1.10" orders before "1.9". Ordering by name would start an
|
|
# older host and call it an upgrade.
|
|
#
|
|
# **A pin stands until something newer arrives.** A version delivered after the pin was written, and
|
|
# never rolled back, is a new build the mesh asks for: the pin goes and it runs, on trial.
|
|
pick_host() {
|
|
newest=
|
|
# A directory with no executable in it is not a version: an interrupted delivery leaves one, and
|
|
# running "the newest" would then mean running nothing.
|
|
for candidate in $(ls -1t "$VERSIONS" 2>/dev/null || true); do
|
|
delivered "$candidate" || continue
|
|
rolled_back "$candidate" && continue
|
|
newest="$candidate"
|
|
break
|
|
done
|
|
if [ -s "$PINNED" ]; then
|
|
pinned="$(tr -d '[:space:]' < "$PINNED" 2>/dev/null || true)"
|
|
if [ -n "$newest" ] && [ "$newest" != "$pinned" ] && [ "$VERSIONS/$newest" -nt "$PINNED" ]; then
|
|
say "host $newest arrived after the rollback to $pinned; running it"
|
|
rm -f "$PINNED"
|
|
elif delivered "$pinned"; then
|
|
echo "$VERSIONS/$pinned/$BINARY"
|
|
return 0
|
|
else
|
|
say "the pinned version '$pinned' is not delivered; ignoring the pin"
|
|
fi
|
|
fi
|
|
if [ -n "$newest" ]; then
|
|
echo "$VERSIONS/$newest/$BINARY"
|
|
return 0
|
|
fi
|
|
echo "$FALLBACK"
|
|
}
|
|
|
|
# Go back from `from` to the known-good version, once, and say so. False when there is nowhere to go.
|
|
roll_back() { # from why
|
|
kg="$(known_good)"
|
|
if [ -z "$1" ] || [ "$1" = "$kg" ] || ! delivered "$kg"; then
|
|
return 1
|
|
fi
|
|
record "$1" "$kg" rolled-back "$2"
|
|
# The pin is what stops the launcher starting a version newer than known-good that is not the one
|
|
# rolled back; the record is what stops it starting this one again.
|
|
printf '%s\n' "$kg" > "$PINNED"
|
|
rm -f "$TRIAL"
|
|
say "rolled back from host $1 to $kg: $2"
|
|
return 0
|
|
}
|
|
|
|
# Watch a host on trial: done when it writes itself into known-good, stopped when the bound passes.
|
|
# A subshell beside the host, so the launcher's own wait is untouched.
|
|
trial_watch() { # pid version deadline
|
|
while kill -0 "$1" 2>/dev/null; do
|
|
[ "$(known_good)" = "$2" ] && return 0
|
|
if [ "$(now)" -ge "$3" ]; then
|
|
printf '%s\n' "$2" > "$EXPIRED"
|
|
say "host $2 did not report within ${BOUND}s of starting; stopping it"
|
|
kill -TERM "$1" 2>/dev/null
|
|
waited=0
|
|
while kill -0 "$1" 2>/dev/null && [ "$waited" -lt "$GRACE" ]; do
|
|
sleep 1
|
|
waited=$((waited + 1))
|
|
done
|
|
kill -KILL "$1" 2>/dev/null
|
|
return 0
|
|
fi
|
|
sleep "$TICK"
|
|
done
|
|
}
|
|
|
|
child=
|
|
watcher=
|
|
stopping=
|
|
|
|
# The machine is shutting down. Pass it on and wait for the host to finish — a supervisor that
|
|
# exits while its child is still running leaves the host to be killed rather than to stop, and
|
|
# an apply interrupted that way is exactly the half-configured machine this project is about.
|
|
on_term() {
|
|
stopping=yes
|
|
if [ -n "$child" ]; then
|
|
say "stopping: passing the signal to the host"
|
|
kill -TERM "$child" 2>/dev/null
|
|
fi
|
|
}
|
|
trap on_term TERM INT
|
|
|
|
mkdir -p "$STATE_DIR"
|
|
# What this launcher is, so a delivered successor of it is run at the next clean exit rather than at
|
|
# the next boot.
|
|
SELF="$(cksum < "$0" 2>/dev/null || true)"
|
|
|
|
while :; do
|
|
if [ -n "$stopping" ]; then
|
|
exit 0
|
|
fi
|
|
|
|
if [ -e "$HALTED" ]; then
|
|
say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')"
|
|
say "not starting the host. this node needs a person."
|
|
exit 0
|
|
fi
|
|
|
|
# Consecutive failed starts, not starts. Cleared by the host itself when it reports, which is
|
|
# the only evidence either this or known-good has.
|
|
#
|
|
# Read the FIRST FIELD, then insist it is a plain integer.
|
|
#
|
|
# Stripping whitespace instead concatenates, and that is not hypothetical: a counter
|
|
# holding "1 2" became "12", past the limit, so a healthy node rolled itself back. An
|
|
# unreadable counter must fail towards "start normally", never towards "give up".
|
|
count=0
|
|
if [ -s "$ATTEMPTS" ]; then
|
|
read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0
|
|
fi
|
|
case "${count:-}" in
|
|
'' | *[!0-9]*) count=0 ;;
|
|
esac
|
|
|
|
HOST="$(pick_host)"
|
|
version="$(version_of "$HOST")"
|
|
|
|
if [ "$count" -ge "$LIMIT" ]; then
|
|
if roll_back "$version" "it failed $count times in a row"; then
|
|
# Fresh count for the version now run: it deserves its own attempts, and without this
|
|
# it inherits a count already over the limit and halts at once. The variable too, not
|
|
# only the file: resetting one and not the other made the next failure count from the
|
|
# OLD value — so the rolled-back version got one attempt instead of three.
|
|
count=0
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
continue
|
|
fi
|
|
kg="$(known_good)"
|
|
if [ -n "$kg" ] && { [ "$version" = "$kg" ] || [ -z "$version" ]; } && [ -s "$ROLLED" ]; then
|
|
# Already gone back, and what it went back to fails as well: this is the machine and
|
|
# not a binary. Rolling back again would flap between two versions for ever.
|
|
say "the host failed $count times after a rollback. the version it went back to does"
|
|
say "not start either, so this is the machine and not the binary."
|
|
record "${version:-placed-by-hand}" "" halted "the version rolled back to failed $count times as well"
|
|
printf 'rolled back and still failing\n' > "$HALTED"
|
|
exit 0
|
|
fi
|
|
# Nothing to go back to: no host has ever reported here, or the one that did was placed by
|
|
# hand and is not delivered. An installation failure rather than an upgrade's — said, and
|
|
# tried again slowly, never guessed at.
|
|
say "the host failed $count times and there is no delivered known-good version to go back to."
|
|
count=0
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
fi
|
|
|
|
if [ ! -x "$HOST" ]; then
|
|
say "no host to run: nothing delivered under $VERSIONS and $FALLBACK is not executable."
|
|
printf 'no host binary\n' > "$HALTED"
|
|
exit 0
|
|
fi
|
|
|
|
# **On trial**: a delivered version that is not the known-good one, while a known-good one is
|
|
# delivered to go back to. The trial starts when this version was first started, and survives
|
|
# this launcher restarting.
|
|
trial=
|
|
deadline=
|
|
kg="$(known_good)"
|
|
if [ -n "$version" ] && [ "$version" != "$kg" ] && delivered "$kg"; then
|
|
trial=yes
|
|
started=
|
|
if [ -s "$TRIAL" ]; then
|
|
read -r on since _ < "$TRIAL" 2>/dev/null || on=
|
|
[ "${on:-}" = "$version" ] && started="${since:-}"
|
|
fi
|
|
case "$started" in
|
|
'' | *[!0-9]*) started="$(now)"; printf '%s %s\n' "$version" "$started" > "$TRIAL" ;;
|
|
esac
|
|
deadline=$((started + BOUND))
|
|
if [ "$(now)" -ge "$deadline" ]; then
|
|
roll_back "$version" "it did not report within ${BOUND}s of starting" && continue
|
|
fi
|
|
else
|
|
rm -f "$TRIAL"
|
|
fi
|
|
|
|
rm -f "$EXPIRED"
|
|
say "running $HOST${trial:+ (on trial until it reports)}"
|
|
"$HOST" run &
|
|
child=$!
|
|
watcher=
|
|
if [ -n "$trial" ]; then
|
|
trial_watch "$child" "$version" "$deadline" &
|
|
watcher=$!
|
|
fi
|
|
status=0
|
|
wait "$child" || status=$?
|
|
if [ -n "$stopping" ]; then
|
|
# The signal interrupted the wait, not the host: it was told, and is finishing what it was
|
|
# doing. Waited for, so the service manager does not kill it half way through an apply.
|
|
wait "$child" 2>/dev/null
|
|
fi
|
|
child=
|
|
if [ -n "$watcher" ]; then
|
|
kill "$watcher" 2>/dev/null
|
|
wait "$watcher" 2>/dev/null
|
|
watcher=
|
|
fi
|
|
|
|
if [ -n "$stopping" ]; then
|
|
exit 0
|
|
fi
|
|
|
|
if [ -n "$trial" ] && [ -s "$EXPIRED" ] && [ "$(cat "$EXPIRED" 2>/dev/null)" = "$version" ]; then
|
|
rm -f "$EXPIRED"
|
|
roll_back "$version" "it did not report within ${BOUND}s of starting"
|
|
count=0
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
continue
|
|
fi
|
|
|
|
# A signal the host did not survive, and we are not shutting down: treat it as a crash.
|
|
case "$status" in
|
|
0)
|
|
# Exited cleanly. That is how the host stands aside for a new binary after an
|
|
# upgrade (novox/hq ADR 0005) — so loop and run whatever is now on disk.
|
|
#
|
|
# Deliberately NOT counted, and this is the whole reason the counter is
|
|
# incremented here rather than before the start: counting attempts meant a host
|
|
# that upgraded itself three times rolled itself back, having worked perfectly
|
|
# every time.
|
|
#
|
|
# **Unless it stood aside for nothing.** A host on trial that exits cleanly while the
|
|
# same host is still the one to run has not stood aside for a successor: it has
|
|
# stopped, and a host that stops at once would otherwise loop here for ever unseen.
|
|
if [ -n "$trial" ] && [ "$(pick_host)" = "$HOST" ] && [ "$(known_good)" != "$version" ]; then
|
|
say "host $version exited cleanly on trial with nothing newer to stand aside for"
|
|
else
|
|
# A delivered successor of this launcher runs from here on, not from the next boot.
|
|
if [ -n "$SELF" ] && [ "$(cksum < "$0" 2>/dev/null || true)" != "$SELF" ]; then
|
|
say "the launcher was replaced; running the new one"
|
|
exec "$0"
|
|
fi
|
|
say "the host exited cleanly; starting it again"
|
|
continue
|
|
fi
|
|
;;
|
|
esac
|
|
|
|
count=$((count + 1))
|
|
printf '%s\n' "$count" > "$ATTEMPTS"
|
|
|
|
say "the host exited $status ($count consecutive); restarting in ${BACKOFF}s"
|
|
[ -n "$ONCE" ] && exit "$status"
|
|
sleep "$BACKOFF"
|
|
done
|