#!/bin/sh # Supervise the host: start it, watch it, and decide what to do when it stops. # # novox/hq ADR 0005. The init is asked for ONE thing — run this at boot — and everything else # lives here, in a script that can be tested. Whether to restart, how long to wait, when to give # up, when to roll back: all of it is policy, and policy in a unit file can only be read and # hoped for. # # **It is the witness of the host's own successor** (novox/hq to-be 45 §8, ADR 0227 rule 8). A # delivered host that is not the known-good one runs on trial: it must report its declaration # under its own build — which it marks by writing itself into known-good — within a bound. One # that crashes repeatedly, exits without standing aside for anything, or does not report in # bound is stopped, recorded in `rolled-back` with why, and the known-good one is run. The host # says every record on its reports, so the mesh raises it as a condition. One rollback per # version: a version in `rolled-back` is never started again; a newer delivery is. # # It does NOT exec the host. Exec would replace this process, and then only the init could # restart anything — which is the arrangement this exists to remove. The cost of staying is # signal handling, below. # # Everything a rollback does is one of: read a file, look at a directory, write a file. It shares # no code with the host and calls none of it: a binary that cannot start cannot be its own # recovery. # # POSIX sh. `set -e` is deliberately absent: this script's whole job is to inspect exit codes, # and -e would make it exit on the first one it is meant to handle. set -u STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}" LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}" # The host that was placed by hand, used only when nothing has been delivered. The first host on a # machine always arrives this way; every one after it is delivered (novox/hq ADR 0141). FALLBACK="${MESH_HOST_BIN:-/usr/bin/nox-mesh-host}" VERSIONS="$LIBEXEC/versions" BINARY="nox-mesh-host" LIMIT="${MESH_HOST_START_LIMIT:-3}" BACKOFF="${MESH_HOST_BACKOFF:-5}" # How long a host on trial has to report, in seconds: to-be 45 §8's ten minutes from the apply. BOUND="${MESH_HOST_TRIAL_BOUND:-600}" TICK="${MESH_HOST_TRIAL_TICK:-5}" GRACE="${MESH_HOST_STOP_GRACE:-20}" ONCE="${MESH_HOST_RUN_ONCE:-}" # tests run one iteration; nothing else sets this ATTEMPTS="$STATE_DIR/start-attempts" HALTED="$STATE_DIR/halted" PINNED="$STATE_DIR/rollback-pinned" KNOWN_GOOD="$STATE_DIR/known-good" # One line per verdict: from, to, when (seconds since the epoch), outcome, why — tab-separated, read # by the host (internal/upgrade) and said on its reports. ROLLED="$STATE_DIR/rolled-back" TRIAL="$STATE_DIR/trial" EXPIRED="$STATE_DIR/trial-expired" say() { echo "nox-mesh-host-launch: $*" >&2; } now() { date +%s; } known_good() { tr -d '[:space:]' < "$KNOWN_GOOD" 2>/dev/null || true; } delivered() { [ -n "$1" ] && [ -x "$VERSIONS/$1/$BINARY" ]; } rolled_back() { [ -s "$ROLLED" ] && cut -f1 "$ROLLED" 2>/dev/null | grep -qxF "$1"; } # The version a host path is: the directory it was delivered in, or nothing for the host placed by hand. version_of() { case "$1" in "$VERSIONS"/*/"$BINARY") v="${1#"$VERSIONS"/}"; echo "${v%/"$BINARY"}" ;; *) echo "" ;; esac } record() { # from to outcome why printf '%s\t%s\t%s\t%s\t%s\n' "$1" "$2" "$(now)" "$3" "$4" >> "$ROLLED" } # Which host to run: the newest delivered one that was never rolled back — or, while a rollback's pin # stands, the version it pinned — or the one placed by hand when nothing has been delivered. # # **Asked every time round the loop, not once.** Standing aside for a successor is a clean exit, and # the next turn has to run what is on disk NOW — resolving this once would restart the same binary # for ever and the upgrade would never take. # # Newest by when it arrived, never by how its name sorts: a version string is whatever the source was # described as, and those do not sort — "1.10" orders before "1.9". Ordering by name would start an # older host and call it an upgrade. # # **A pin stands until something newer arrives.** A version delivered after the pin was written, and # never rolled back, is a new build the mesh asks for: the pin goes and it runs, on trial. pick_host() { newest= # A directory with no executable in it is not a version: an interrupted delivery leaves one, and # running "the newest" would then mean running nothing. for candidate in $(ls -1t "$VERSIONS" 2>/dev/null || true); do delivered "$candidate" || continue rolled_back "$candidate" && continue newest="$candidate" break done if [ -s "$PINNED" ]; then pinned="$(tr -d '[:space:]' < "$PINNED" 2>/dev/null || true)" if [ -n "$newest" ] && [ "$newest" != "$pinned" ] && [ "$VERSIONS/$newest" -nt "$PINNED" ]; then say "host $newest arrived after the rollback to $pinned; running it" rm -f "$PINNED" elif delivered "$pinned"; then echo "$VERSIONS/$pinned/$BINARY" return 0 else say "the pinned version '$pinned' is not delivered; ignoring the pin" fi fi if [ -n "$newest" ]; then echo "$VERSIONS/$newest/$BINARY" return 0 fi echo "$FALLBACK" } # Go back from `from` to the known-good version, once, and say so. False when there is nowhere to go. roll_back() { # from why kg="$(known_good)" if [ -z "$1" ] || [ "$1" = "$kg" ] || ! delivered "$kg"; then return 1 fi record "$1" "$kg" rolled-back "$2" # The pin is what stops the launcher starting a version newer than known-good that is not the one # rolled back; the record is what stops it starting this one again. printf '%s\n' "$kg" > "$PINNED" rm -f "$TRIAL" say "rolled back from host $1 to $kg: $2" return 0 } # Watch a host on trial: done when it writes itself into known-good, stopped when the bound passes. # A subshell beside the host, so the launcher's own wait is untouched. trial_watch() { # pid version deadline while kill -0 "$1" 2>/dev/null; do [ "$(known_good)" = "$2" ] && return 0 if [ "$(now)" -ge "$3" ]; then printf '%s\n' "$2" > "$EXPIRED" say "host $2 did not report within ${BOUND}s of starting; stopping it" kill -TERM "$1" 2>/dev/null waited=0 while kill -0 "$1" 2>/dev/null && [ "$waited" -lt "$GRACE" ]; do sleep 1 waited=$((waited + 1)) done kill -KILL "$1" 2>/dev/null return 0 fi sleep "$TICK" done } child= watcher= stopping= # The machine is shutting down. Pass it on and wait for the host to finish — a supervisor that # exits while its child is still running leaves the host to be killed rather than to stop, and # an apply interrupted that way is exactly the half-configured machine this project is about. on_term() { stopping=yes if [ -n "$child" ]; then say "stopping: passing the signal to the host" kill -TERM "$child" 2>/dev/null fi } trap on_term TERM INT mkdir -p "$STATE_DIR" # What this launcher is, so a delivered successor of it is run at the next clean exit rather than at # the next boot. SELF="$(cksum < "$0" 2>/dev/null || true)" while :; do if [ -n "$stopping" ]; then exit 0 fi if [ -e "$HALTED" ]; then say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')" say "not starting the host. this node needs a person." exit 0 fi # Consecutive failed starts, not starts. Cleared by the host itself when it reports, which is # the only evidence either this or known-good has. # # Read the FIRST FIELD, then insist it is a plain integer. # # Stripping whitespace instead concatenates, and that is not hypothetical: a counter # holding "1 2" became "12", past the limit, so a healthy node rolled itself back. An # unreadable counter must fail towards "start normally", never towards "give up". count=0 if [ -s "$ATTEMPTS" ]; then read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0 fi case "${count:-}" in '' | *[!0-9]*) count=0 ;; esac HOST="$(pick_host)" version="$(version_of "$HOST")" if [ "$count" -ge "$LIMIT" ]; then if roll_back "$version" "it failed $count times in a row"; then # Fresh count for the version now run: it deserves its own attempts, and without this # it inherits a count already over the limit and halts at once. The variable too, not # only the file: resetting one and not the other made the next failure count from the # OLD value — so the rolled-back version got one attempt instead of three. count=0 printf '%s\n' "$count" > "$ATTEMPTS" continue fi kg="$(known_good)" if [ -n "$kg" ] && { [ "$version" = "$kg" ] || [ -z "$version" ]; } && [ -s "$ROLLED" ]; then # Already gone back, and what it went back to fails as well: this is the machine and # not a binary. Rolling back again would flap between two versions for ever. say "the host failed $count times after a rollback. the version it went back to does" say "not start either, so this is the machine and not the binary." record "${version:-placed-by-hand}" "" halted "the version rolled back to failed $count times as well" printf 'rolled back and still failing\n' > "$HALTED" exit 0 fi # Nothing to go back to: no host has ever reported here, or the one that did was placed by # hand and is not delivered. An installation failure rather than an upgrade's — said, and # tried again slowly, never guessed at. say "the host failed $count times and there is no delivered known-good version to go back to." count=0 printf '%s\n' "$count" > "$ATTEMPTS" fi if [ ! -x "$HOST" ]; then say "no host to run: nothing delivered under $VERSIONS and $FALLBACK is not executable." printf 'no host binary\n' > "$HALTED" exit 0 fi # **On trial**: a delivered version that is not the known-good one, while a known-good one is # delivered to go back to. The trial starts when this version was first started, and survives # this launcher restarting. trial= deadline= kg="$(known_good)" if [ -n "$version" ] && [ "$version" != "$kg" ] && delivered "$kg"; then trial=yes started= if [ -s "$TRIAL" ]; then read -r on since _ < "$TRIAL" 2>/dev/null || on= [ "${on:-}" = "$version" ] && started="${since:-}" fi case "$started" in '' | *[!0-9]*) started="$(now)"; printf '%s %s\n' "$version" "$started" > "$TRIAL" ;; esac deadline=$((started + BOUND)) if [ "$(now)" -ge "$deadline" ]; then roll_back "$version" "it did not report within ${BOUND}s of starting" && continue fi else rm -f "$TRIAL" fi rm -f "$EXPIRED" say "running $HOST${trial:+ (on trial until it reports)}" "$HOST" run & child=$! watcher= if [ -n "$trial" ]; then trial_watch "$child" "$version" "$deadline" & watcher=$! fi status=0 wait "$child" || status=$? if [ -n "$stopping" ]; then # The signal interrupted the wait, not the host: it was told, and is finishing what it was # doing. Waited for, so the service manager does not kill it half way through an apply. wait "$child" 2>/dev/null fi child= if [ -n "$watcher" ]; then kill "$watcher" 2>/dev/null wait "$watcher" 2>/dev/null watcher= fi if [ -n "$stopping" ]; then exit 0 fi if [ -n "$trial" ] && [ -s "$EXPIRED" ] && [ "$(cat "$EXPIRED" 2>/dev/null)" = "$version" ]; then rm -f "$EXPIRED" roll_back "$version" "it did not report within ${BOUND}s of starting" count=0 printf '%s\n' "$count" > "$ATTEMPTS" continue fi # A signal the host did not survive, and we are not shutting down: treat it as a crash. case "$status" in 0) # Exited cleanly. That is how the host stands aside for a new binary after an # upgrade (novox/hq ADR 0005) — so loop and run whatever is now on disk. # # Deliberately NOT counted, and this is the whole reason the counter is # incremented here rather than before the start: counting attempts meant a host # that upgraded itself three times rolled itself back, having worked perfectly # every time. # # **Unless it stood aside for nothing.** A host on trial that exits cleanly while the # same host is still the one to run has not stood aside for a successor: it has # stopped, and a host that stops at once would otherwise loop here for ever unseen. if [ -n "$trial" ] && [ "$(pick_host)" = "$HOST" ] && [ "$(known_good)" != "$version" ]; then say "host $version exited cleanly on trial with nothing newer to stand aside for" else # A delivered successor of this launcher runs from here on, not from the next boot. if [ -n "$SELF" ] && [ "$(cksum < "$0" 2>/dev/null || true)" != "$SELF" ]; then say "the launcher was replaced; running the new one" exec "$0" fi say "the host exited cleanly; starting it again" continue fi ;; esac count=$((count + 1)) printf '%s\n' "$count" > "$ATTEMPTS" say "the host exited $status ($count consecutive); restarting in ${BACKOFF}s" [ -n "$ONCE" ] && exit "$status" sleep "$BACKOFF" done