#!/bin/sh # nats's entrypoint: run the server, and reload it in place when the mesh rewrites its # configuration. # # **Why this exists inside the module** (novox/hq design 25 §5). The controller composes every # account and permission into one file, and that file changes whenever a module is added, # reassigned, or a person's access is granted or revoked — which is often, and on the one server # everything else depends on. The host has no way to say "reload this container": a # container resource has `restart-on` and nothing else, and a container's `restart-on` means # *recreate* — every connection dropped and every in-flight JetStream ack lost, mid-flight, for a # permission change. `reload-on` is real but it is a *service* field, not a container's. # # nats-server already reloads its own configuration on SIGHUP — accounts, permissions, everything # the mesh composes — without dropping a connection. That is the server's own documented # capability, not something built for the mesh. So the configuration is mounted as a directory # (a directory's contents are not digest-tracked the way a directly-mounted file's are, novox/hq # issue 103), and this watches the one file inside it and signals the server itself. The host's # only job is what it already does for any directory: keep the file's content current. Nothing # here is declared `restart-on` or `reload-on`. set -eu # **Two files, and only one of them is the mesh's** (novox/hq design 25 §4, task 1.7). CONF is this # module's own — ports, TLS, JetStream — declared in its manifest, because those are properties of # the container this module raises. USERS is every account and permission, composed by the # controller, and CONF includes it. So what is watched here is the mesh's half: the module's own # does not change without a new declaration, and that recreates the container anyway. CONF="${MESH_NATS_CONF:-/etc/nats/nats.conf}" USERS="${MESH_NATS_USERS:-/etc/nats/accounts.conf}" POLL="${MESH_NATS_CONF_POLL_SECONDS:-5}" # Both are written as part of the same declaration that creates this container, but none of the # three are ordered against each other. Waiting is correct and starting without them is not: # nats-server given a configuration whose include is missing refuses to start, and one given no # configuration at all comes up with its compiled-in defaults — no TLS, no accounts, every subject # open to anyone who can reach the port. A bus that is briefly open to everything is not a bus that # is briefly wrong; it is an open bus. for needed in "$CONF" "$USERS"; do while [ ! -s "$needed" ]; do echo "[nats] waiting for the mesh to write $needed" sleep 1 done done digest() { sha256sum "$USERS" 2>/dev/null | cut -d' ' -f1; } nats-server --config "$CONF" "$@" & server=$! # Forward a stop to the server and let it drain, rather than dying and leaving it orphaned as # PID 1's child. stop() { kill -TERM "$server" 2>/dev/null || true; } trap stop TERM INT last=$(digest) while kill -0 "$server" 2>/dev/null; do sleep "$POLL" now=$(digest) # An empty digest means the file is mid-write or briefly gone. Reloading on that would hand the # server a truncated configuration; the next tick sees the finished one. [ -n "$now" ] || continue if [ "$now" != "$last" ]; then last=$now echo "[nats] the mesh's user list changed; reloading in place" kill -HUP "$server" || true fi done # `wait` on an already-exited child still yields its status, which becomes this container's. wait "$server"