#!/bin/sh # Put the host back on the last version that worked. # # novox/hq ADR 0005 and ADR 0141. This runs when nox-mesh-host will not start, so it shares no code # with it and calls none of it: a binary that cannot start cannot be its own recovery. POSIX sh, no # bashisms, nothing that has to be installed. # # It is deliberately dull. Everything it does is one of: read a file, look at a directory, write a # file. # # **It used to reinstall a package.** It read the known-good version and asked one operating system's # package manager for it, out of that package manager's cache. Two things were wrong with that. No # machine in this mesh had the host installed as a package, so the recovery could not run on any of # them; and the host is built per operating system (ADR 0005), so a recovery written in one package # manager's terms could not run on two of the three. Versions now live side by side in directories # named for them, so going back is choosing a directory — which is the same on every machine. set -eu STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}" LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}" VERSIONS="$LIBEXEC/versions" BINARY="nox-mesh-host" KNOWN_GOOD="$STATE_DIR/known-good" ATTEMPTED="$STATE_DIR/rollback-attempted" PINNED="$STATE_DIR/rollback-pinned" say() { echo "nox-mesh-host-rollback: $*" >&2; } # Roll back once. A second failure is a different diagnosis: the previously working binary also # does not run, so the binary is not the problem — the machine is. Rolling back again would flap # between two versions forever and bury the actual cause under a loop. if [ -e "$ATTEMPTED" ]; then say "already rolled back once, to $(cat "$ATTEMPTED" 2>/dev/null || echo unknown)." say "the previous version also failed to start, so this is the machine and not the binary." say "not rolling back again. this node needs a person." exit 0 fi # A machine whose host never completed a reconcile has no version to go back to. That is a real # state rather than a fault: the node was never working, so the failure belongs to the # installation. Guessing a version here is how a recovery becomes a second fault. if [ ! -s "$KNOWN_GOOD" ]; then say "no known-good version recorded — this host has never completed a reconcile." say "there is nothing to roll back to. this is an installation failure, not an upgrade one." exit 0 fi VERSION="$(tr -d '[:space:]' < "$KNOWN_GOOD")" if [ -z "$VERSION" ]; then say "known-good is empty. refusing to guess." exit 0 fi # The version that last worked may be the one that was placed by hand, which is not delivered and has # no directory. Nothing to choose, and saying so is better than pinning a version that is not there — # the launcher would ignore the pin and start the newest again, which is the binary that is failing. if [ ! -x "$VERSIONS/$VERSION/$BINARY" ]; then say "known-good is $VERSION and no such version is delivered under $VERSIONS." say "it was retired, or that host was placed by hand and never delivered." say "cannot roll back. this node needs a person." exit 1 fi say "rolling back to $VERSION ($VERSIONS/$VERSION/$BINARY)" printf '%s\n' "$VERSION" > "$ATTEMPTED" # The pin is what stops the launcher starting the newest again. Written last, so a failure above # leaves the machine choosing for itself rather than pinned to something this script did not verify. printf '%s\n' "$VERSION" > "$PINNED" # Deliberately does NOT start anything. The launcher called this and will run the host next, so # starting it here would run two. novox/hq ADR 0005 moved that responsibility; this script chooses a # version and says so, and nothing else. say "pinned $VERSION. the launcher will start it."