The launcher supervises the host instead of exec'ing it
Jochen: "I thought we did not want to run the host under a systemd/openrc/init loop, but instead had our own host-init program?" -- and that was right. I had moved the give-up logic out of unit files and left RESTART in them, with the launcher exec'ing the host and disappearing. So init still decided when the host came back, which is the arrangement 0061 exists to remove. The launcher now stays and supervises: starts the host as a child, waits, decides. Init is asked for one thing, run this at boot. There is an OpenRC script beside the systemd unit now, four lines each, which is the point -- a second init is transcription rather than a port. The cost of not exec'ing is signals. A supervisor that exits while its child runs leaves the host to be killed rather than to stop, and an apply interrupted that way is the half-configured machine this project is about. So SIGTERM is trapped, passed down, and waited on. Two bugs, both found by the tests rather than by review: A clean exit was counted as a failure. The host exits cleanly to stand aside for a new binary after an upgrade (0057), so a host that upgraded itself three times rolled itself back having worked perfectly every time. The counter now counts CONSECUTIVE FAILURES, incremented after the wait rather than before the start. And when rolling back I reset the counter file but not the variable, so the next failure counted from the old value -- the rolled-back version got one attempt instead of three. Also: the host now clears the counter when it completes a reconcile, at the same moment it records known-good and for the same reason. Without it the count only climbs, and a node up for months rolls itself back on its third ordinary restart -- a healthy machine undone by its own recovery. One test expectation was tightened rather than fixed: "resets the counter after rolling back" asserted exactly 0, which was true only under the old count-before-start semantics. It now asserts the property -- below the limit -- since 1 is correct after a rollback plus one failure. 32 launcher tests, all confirmed to bite.
This commit is contained in:
@@ -327,6 +327,13 @@ func runApply(ctx context.Context, opts options, d *declaration.Declaration, sou
|
||||
" a rollback would have nothing to return to.\n", version, err)
|
||||
}
|
||||
}
|
||||
// And tell the launcher this start worked. Without it the counter only climbs, and a node
|
||||
// that has been up for months rolls itself back on its third ordinary restart.
|
||||
if err := upgrade.ClearAttempts(upgrade.AttemptsPath(opts.state)); err != nil {
|
||||
fmt.Fprintf(os.Stderr,
|
||||
"mesh-host: applied, but could not clear the start counter: %v\n"+
|
||||
" this node may roll itself back after a few more restarts.\n", err)
|
||||
}
|
||||
|
||||
if opts.json {
|
||||
return writeJSON(report)
|
||||
|
||||
@@ -19,9 +19,12 @@ import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
// KnownGoodName is the file a rollback script reads. Next to the store, because it is node
|
||||
// Files the launcher reads and this binary writes. Next to the store, because they are node
|
||||
// state of exactly the same kind.
|
||||
const KnownGoodName = "known-good"
|
||||
const (
|
||||
KnownGoodName = "known-good"
|
||||
AttemptsName = "start-attempts"
|
||||
)
|
||||
|
||||
// Self is the executable this process started from, remembered.
|
||||
//
|
||||
@@ -80,6 +83,24 @@ func KnownGoodPath(statePath string) string {
|
||||
return filepath.Join(filepath.Dir(statePath), KnownGoodName)
|
||||
}
|
||||
|
||||
// AttemptsPath is where the launcher counts starts that have not yet worked.
|
||||
func AttemptsPath(statePath string) string {
|
||||
return filepath.Join(filepath.Dir(statePath), AttemptsName)
|
||||
}
|
||||
|
||||
// ClearAttempts tells the launcher this start worked.
|
||||
//
|
||||
// Written at the same moment as known-good and for the same reason: a completed reconcile is
|
||||
// the evidence, and it is the only evidence either of them has. Without this the counter only
|
||||
// ever climbs, so a node that has been up for months rolls itself back on its third ordinary
|
||||
// restart — a healthy machine undone by its own recovery.
|
||||
func ClearAttempts(path string) error {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(path, []byte("0\n"), 0o644)
|
||||
}
|
||||
|
||||
// RecordKnownGood marks a version as one that started and completed a reconcile.
|
||||
//
|
||||
// Written atomically and as one bare line. The reader is a shell script running on a machine
|
||||
|
||||
@@ -15,6 +15,8 @@ setup() {
|
||||
export MESH_HOST_LIBEXEC="$WORK/libexec"
|
||||
export MESH_HOST_BIN="$WORK/bin/nox-mesh-host"
|
||||
export MESH_HOST_START_LIMIT=3
|
||||
export MESH_HOST_BACKOFF=0
|
||||
export MESH_HOST_RUN_ONCE=1
|
||||
mkdir -p "$MESH_HOST_STATE_DIR" "$MESH_HOST_LIBEXEC" "$WORK/bin"
|
||||
|
||||
# A host that records being started. It exits immediately, which is what the launcher's
|
||||
@@ -23,7 +25,7 @@ setup() {
|
||||
cat > "$MESH_HOST_BIN" <<'STUB'
|
||||
#!/bin/sh
|
||||
echo "$@" >> "$MESH_HOST_STATE_DIR/host.starts"
|
||||
exit 0
|
||||
exit "${STUB_HOST_EXIT:-1}"
|
||||
STUB
|
||||
cat > "$MESH_HOST_LIBEXEC/rollback" <<'STUB'
|
||||
#!/bin/sh
|
||||
@@ -66,8 +68,11 @@ echo "1.4.2" > "$MESH_HOST_STATE_DIR/known-good"
|
||||
i=1; while [ $i -le 4 ]; do "$LAUNCH" >/dev/null 2>&1 || true; i=$((i+1)); done
|
||||
check "the fourth start rolls back" "three failures is a binary that does not work" "$(rolled)" "yes"
|
||||
check "and still starts the host" "the rolled-back version has to be run" "$(started)" "yes"
|
||||
# Below the limit, not exactly zero. The rollback resets it and the rolled-back version then
|
||||
# fails once here, so 1 is right — the property is that it did NOT inherit a count already at
|
||||
# the limit, which would halt the new version on its first attempt.
|
||||
check "resets the counter after rolling back" "the new version deserves its own attempts, or it halts at once" \
|
||||
"$(count)" "0"
|
||||
"$([ "$(count)" -lt 3 ] && echo below-limit || echo "at-limit($(count))")" "below-limit"
|
||||
|
||||
# --- the host clears the counter on success ----------------------------------------------------
|
||||
setup
|
||||
@@ -131,5 +136,56 @@ for corrupt in "5x" "0x10" "1 2" ""; do
|
||||
"$(count | grep -cE '^[0-9]+$')" "1"
|
||||
done
|
||||
|
||||
# --- the loop, and shutting down ------------------------------------------------------------
|
||||
#
|
||||
# These need the launcher to actually run as a supervisor rather than one iteration, so they do
|
||||
# not set MESH_HOST_RUN_ONCE.
|
||||
|
||||
# A host that exits 0 has upgraded itself and stood aside (novox/hq ADR 0057). The launcher must
|
||||
# start it again — and must NOT count it, because it did not fail.
|
||||
setup
|
||||
unset MESH_HOST_RUN_ONCE
|
||||
cat > "$MESH_HOST_BIN" <<'STUB'
|
||||
#!/bin/sh
|
||||
echo start >> "$MESH_HOST_STATE_DIR/host.starts"
|
||||
# Exit 0 three times, then hang so the launcher stops looping and can be killed.
|
||||
if [ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts")" -lt 3 ]; then exit 0; fi
|
||||
sleep 30
|
||||
STUB
|
||||
chmod +x "$MESH_HOST_BIN"
|
||||
"$LAUNCH" >/dev/null 2>&1 &
|
||||
LP=$!
|
||||
sleep 1
|
||||
check "a clean exit restarts the host" "that is how it stands aside for a new binary" \
|
||||
"$([ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts" 2>/dev/null || echo 0)" -ge 3 ] && echo looped || echo stopped)" "looped"
|
||||
# No counter file at all: nothing has failed, so nothing has been counted.
|
||||
check "a clean exit is not counted as a failure" "it finished, it did not fail" "$(count)" "MISSING"
|
||||
|
||||
# Shutting down: the signal must reach the host, and the launcher must wait for it rather than
|
||||
# exiting and leaving the host to be killed mid-apply.
|
||||
kill -TERM "$LP" 2>/dev/null
|
||||
sleep 1
|
||||
check "SIGTERM stops the launcher" "a supervisor that ignores shutdown hangs the machine" \
|
||||
"$(kill -0 "$LP" 2>/dev/null && echo running || echo stopped)" "stopped"
|
||||
check "and does not leave the host running" "the child must go down with it" \
|
||||
"$(pgrep -f "$MESH_HOST_BIN" >/dev/null 2>&1 && echo orphaned || echo reaped)" "reaped"
|
||||
|
||||
# A crash IS counted, and the launcher keeps going.
|
||||
setup
|
||||
unset MESH_HOST_RUN_ONCE
|
||||
export MESH_HOST_BACKOFF=0
|
||||
cat > "$MESH_HOST_BIN" <<'STUB'
|
||||
#!/bin/sh
|
||||
echo start >> "$MESH_HOST_STATE_DIR/host.starts"
|
||||
if [ "$(wc -l < "$MESH_HOST_STATE_DIR/host.starts")" -lt 2 ]; then exit 3; fi
|
||||
sleep 30
|
||||
STUB
|
||||
chmod +x "$MESH_HOST_BIN"
|
||||
"$LAUNCH" >/dev/null 2>&1 &
|
||||
LP=$!
|
||||
sleep 1
|
||||
check "a crash is counted" "unlike a clean exit, which is not" "$(count)" "1"
|
||||
kill -TERM "$LP" 2>/dev/null; sleep 1; pkill -f "$MESH_HOST_BIN" 2>/dev/null || true
|
||||
|
||||
printf '\nlaunch: %d passed, %d failed\n' "$PASS" "$FAIL"
|
||||
[ "$FAIL" -eq 0 ]
|
||||
|
||||
+107
-50
@@ -1,72 +1,129 @@
|
||||
#!/bin/sh
|
||||
# Start the host, and decide what to do when it will not start.
|
||||
# Supervise the host: start it, watch it, and decide what to do when it stops.
|
||||
#
|
||||
# novox/hq ADR 0061. The init is asked for two things — start this at boot, start it again if it
|
||||
# exits — and everything else is here, because this is the one piece that has to work on a
|
||||
# machine where the host does not. Unit-file syntax cannot be tested; this can.
|
||||
# novox/hq ADR 0061. The init is asked for ONE thing — run this at boot — and everything else
|
||||
# lives here, in a script that can be tested. Whether to restart, how long to wait, when to give
|
||||
# up, when to roll back: all of it is policy, and policy in a unit file can only be read and
|
||||
# hoped for.
|
||||
#
|
||||
# POSIX sh, no bashisms, nothing that has to be installed.
|
||||
set -eu
|
||||
# It does NOT exec the host. Exec would replace this process, and then only the init could
|
||||
# restart anything — which is the arrangement this exists to remove. The cost of staying is
|
||||
# signal handling, below.
|
||||
#
|
||||
# POSIX sh. `set -e` is deliberately absent: this script's whole job is to inspect exit codes,
|
||||
# and -e would make it exit on the first one it is meant to handle.
|
||||
set -u
|
||||
|
||||
STATE_DIR="${MESH_HOST_STATE_DIR:-/var/lib/mesh-host}"
|
||||
LIBEXEC="${MESH_HOST_LIBEXEC:-/usr/lib/nox-mesh-host}"
|
||||
HOST="${MESH_HOST_BIN:-/usr/bin/nox-mesh-host}"
|
||||
LIMIT="${MESH_HOST_START_LIMIT:-3}"
|
||||
BACKOFF="${MESH_HOST_BACKOFF:-5}"
|
||||
ONCE="${MESH_HOST_RUN_ONCE:-}" # tests run one iteration; nothing else sets this
|
||||
|
||||
ATTEMPTS="$STATE_DIR/start-attempts"
|
||||
HALTED="$STATE_DIR/halted"
|
||||
|
||||
say() { echo "nox-mesh-host-launch: $*" >&2; }
|
||||
|
||||
child=
|
||||
stopping=
|
||||
|
||||
# The machine is shutting down. Pass it on and wait for the host to finish — a supervisor that
|
||||
# exits while its child is still running leaves the host to be killed rather than to stop, and
|
||||
# an apply interrupted that way is exactly the half-configured machine this project is about.
|
||||
on_term() {
|
||||
stopping=yes
|
||||
if [ -n "$child" ]; then
|
||||
say "stopping: passing the signal to the host"
|
||||
kill -TERM "$child" 2>/dev/null
|
||||
fi
|
||||
}
|
||||
trap on_term TERM INT
|
||||
|
||||
mkdir -p "$STATE_DIR"
|
||||
|
||||
# Halted: rolled back once and the previous version failed too, so the binary is not the problem.
|
||||
# Nothing further is tried automatically. Exit zero — a supervisor restarting this forever is a
|
||||
# slow visible loop rather than a crash loop, and the node stays down until a person looks.
|
||||
if [ -e "$HALTED" ]; then
|
||||
say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')"
|
||||
say "not starting the host. this node needs a person."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Read the FIRST FIELD, then insist it is a plain integer.
|
||||
#
|
||||
# Stripping whitespace instead concatenates, and that is not a hypothetical: a counter file
|
||||
# holding "1 2" became "12", which is past the limit, so a healthy node rolled itself back. An
|
||||
# unreadable counter must fail towards "start normally", never towards "give up".
|
||||
count=0
|
||||
if [ -s "$ATTEMPTS" ]; then
|
||||
read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0
|
||||
fi
|
||||
case "${count:-}" in
|
||||
'' | *[!0-9]*) count=0 ;;
|
||||
esac
|
||||
|
||||
count=$((count + 1))
|
||||
printf '%s\n' "$count" > "$ATTEMPTS"
|
||||
|
||||
if [ "$count" -gt "$LIMIT" ]; then
|
||||
# The host has failed to get through a reconcile $LIMIT times running. The counter is
|
||||
# cleared by the host itself on success, so reaching here means none of those starts
|
||||
# worked — not that the machine has been up a long time.
|
||||
if [ -e "$STATE_DIR/rollback-attempted" ]; then
|
||||
say "the host failed $count times after a rollback. the previous version does not"
|
||||
say "start either, so this is the machine and not the binary."
|
||||
printf 'rolled back and still failing\n' > "$HALTED"
|
||||
while :; do
|
||||
if [ -n "$stopping" ]; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
say "the host failed $count times. rolling back."
|
||||
if "$LIBEXEC/rollback"; then
|
||||
# Fresh count for the version we just installed: it deserves its own attempts, and
|
||||
# without this it inherits a count already over the limit and halts immediately.
|
||||
printf '0\n' > "$ATTEMPTS"
|
||||
else
|
||||
say "rollback failed. halting rather than restarting into the same failure."
|
||||
printf 'rollback failed\n' > "$HALTED"
|
||||
if [ -e "$HALTED" ]; then
|
||||
say "halted: $(cat "$HALTED" 2>/dev/null || echo 'reason not recorded')"
|
||||
say "not starting the host. this node needs a person."
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
# exec, so the host is what the supervisor watches and signals reach it directly.
|
||||
exec "$HOST" run
|
||||
# Consecutive failed starts, not starts. Cleared by the host itself when it completes a
|
||||
# reconcile, which is the only evidence either this or known-good has.
|
||||
#
|
||||
# Read the FIRST FIELD, then insist it is a plain integer.
|
||||
#
|
||||
# Stripping whitespace instead concatenates, and that is not hypothetical: a counter
|
||||
# holding "1 2" became "12", past the limit, so a healthy node rolled itself back. An
|
||||
# unreadable counter must fail towards "start normally", never towards "give up".
|
||||
count=0
|
||||
if [ -s "$ATTEMPTS" ]; then
|
||||
read -r count _ < "$ATTEMPTS" 2>/dev/null || count=0
|
||||
fi
|
||||
case "${count:-}" in
|
||||
'' | *[!0-9]*) count=0 ;;
|
||||
esac
|
||||
|
||||
if [ "$count" -ge "$LIMIT" ]; then
|
||||
if [ -e "$STATE_DIR/rollback-attempted" ]; then
|
||||
say "the host failed $count times after a rollback. the previous version does not"
|
||||
say "start either, so this is the machine and not the binary."
|
||||
printf 'rolled back and still failing\n' > "$HALTED"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
say "the host failed $count times. rolling back."
|
||||
if "$LIBEXEC/rollback"; then
|
||||
# Fresh count for the version just installed: it deserves its own attempts, and
|
||||
# without this it inherits a count already over the limit and halts at once.
|
||||
#
|
||||
# The variable too, not only the file. Resetting one and not the other made the
|
||||
# next failure count from the OLD value — so the rolled-back version got one
|
||||
# attempt instead of three.
|
||||
count=0
|
||||
printf '%s\n' "$count" > "$ATTEMPTS"
|
||||
else
|
||||
say "rollback failed. halting rather than restarting into the same failure."
|
||||
printf 'rollback failed\n' > "$HALTED"
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
"$HOST" run &
|
||||
child=$!
|
||||
status=0
|
||||
wait "$child" || status=$?
|
||||
child=
|
||||
|
||||
if [ -n "$stopping" ]; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# A signal the host did not survive, and we are not shutting down: treat it as a crash.
|
||||
case "$status" in
|
||||
0)
|
||||
# Exited cleanly. That is how the host stands aside for a new binary after an
|
||||
# upgrade (novox/hq ADR 0057) — so loop and run whatever is now on disk.
|
||||
#
|
||||
# Deliberately NOT counted, and this is the whole reason the counter is
|
||||
# incremented here rather than before the start: counting attempts meant a host
|
||||
# that upgraded itself three times rolled itself back, having worked perfectly
|
||||
# every time.
|
||||
say "the host exited cleanly; starting it again"
|
||||
continue
|
||||
;;
|
||||
esac
|
||||
|
||||
count=$((count + 1))
|
||||
printf '%s\n' "$count" > "$ATTEMPTS"
|
||||
|
||||
say "the host exited $status ($count consecutive); restarting in ${BACKOFF}s"
|
||||
[ -n "$ONCE" ] && exit "$status"
|
||||
sleep "$BACKOFF"
|
||||
done
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
#!/sbin/openrc-run
|
||||
# The Alpine equivalent of the systemd unit beside this. Four lines of the same two facts:
|
||||
# run the launcher, and bring it back if it dies. Everything else is in the launcher, which is
|
||||
# what makes a second init transcription rather than a port (novox/hq ADR 0061).
|
||||
name="nox-mesh-host"
|
||||
command="/usr/lib/nox-mesh-host/launch"
|
||||
supervisor="supervise-daemon"
|
||||
depend() { need net; }
|
||||
@@ -3,16 +3,15 @@ Description=Novox Mesh node host
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
# Two lines of policy and no more (novox/hq ADR 0061). Counting failed starts and rolling back
|
||||
# lives in the launcher, where it can be tested — so this file is transcription for any other
|
||||
# init rather than design.
|
||||
# One line of policy: run the launcher at boot (novox/hq ADR 0061). Restarting the host,
|
||||
# backing off, giving up and rolling back are all the launcher's, where they can be tested.
|
||||
# Restart= here is a backstop for the launcher itself being killed, not the mechanism.
|
||||
[Service]
|
||||
ExecStart=/usr/lib/nox-mesh-host/launch
|
||||
# always, NOT on-failure: the host restarts onto a new binary by exiting CLEANLY
|
||||
# (novox/hq ADR 0057), and on-failure would leave every upgraded node stopped.
|
||||
Restart=always
|
||||
RestartSec=5s
|
||||
StateDirectory=mesh-host
|
||||
KillMode=mixed
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
|
||||
Reference in New Issue
Block a user