One kind for the module's own code, with three modes
The first cut of this added a `daemon` for the long-running case alone. That would have meant a new vocabulary entry for each of the others — a scheduled task, a run-once migration, a health check — when they are one thing run at different cadences. That is a field, not four entries in a vocabulary where every entry widens what a compromised control plane can express. So it mirrors a container exactly, because it IS a container's twin: the same intent, hosted by the machine's own supervisor instead of a runtime. Stays up, runs once, or runs on a schedule. Tools, hooks and event consumers are not further modes. They are loaded by a tool host, which is itself a process that stays up — so the generic case already covers them, which is the test of whether it is generic. A scheduled process gets a timer and a unit that finishes; a long-running one gets a unit that is restarted when it exits. Getting that wrong either way is a second copy running continuously between fires, or a schedule that never fires. The modes are exclusive and validation says so near the author: something that runs once does not run on a schedule, and something not running between fires cannot be restarted when a file changes. A missed fire happens when the machine comes back rather than being skipped, which is the difference between a machine that was down and a schedule that quietly stopped. Claude-Session: https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx
This commit is contained in:
@@ -0,0 +1,266 @@
|
||||
package apply
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"github.com/novox/mesh-host/internal/declaration"
|
||||
"github.com/novox/mesh-host/internal/store"
|
||||
)
|
||||
|
||||
// Running the mesh's own code, without the module choosing how.
|
||||
//
|
||||
// **A daemon is an intent and this is one answer to it.** A module says what to run and the
|
||||
// machine's own supervisor is how — which means the module is not writing a unit file, and not
|
||||
// choosing between a container and a service before it can declare anything
|
||||
// (novox/hq 03-DESIGN/01-to-be/18-building-a-module.md).
|
||||
//
|
||||
// What this does, in order: fetch the bundle, refuse it unless it hashes to what was declared,
|
||||
// unpack it where the mesh keeps such things, write the unit, and put it in the state asked for.
|
||||
// The unit is the mesh's — an operator editing it loses the edit at the next declaration, which is
|
||||
// the same rule every managed file on a machine follows (ADR 0011).
|
||||
|
||||
// daemonRoot is where unpacked daemons live.
|
||||
//
|
||||
// Under the mesh's own directory rather than somewhere a distribution owns: these are files the
|
||||
// mesh puts there and replaces, and putting them where a package manager also writes is how two
|
||||
// owners end up disagreeing about one path.
|
||||
const daemonRoot = "/var/lib/mesh/daemons"
|
||||
|
||||
// unitDir is where the mesh writes the units it owns.
|
||||
const unitDir = "/etc/systemd/system"
|
||||
|
||||
func applyProcess(ctx context.Context, r *declaration.Process, run Runner,
|
||||
changed map[string]bool, previous store.Applied) (Outcome, error) {
|
||||
out := begin(r)
|
||||
out.Action = "unchanged"
|
||||
|
||||
body, err := fetch(ctx, r.Source)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
sum := sha256.Sum256(body)
|
||||
got := "sha256:" + hex.EncodeToString(sum[:])
|
||||
if got != r.Digest {
|
||||
// Refused before anything is written or started. What is at that address is not what was
|
||||
// declared, and running it would be running something nobody reviewed.
|
||||
return out, fmt.Errorf(
|
||||
"%s was declared as %s and what arrived is %s; nothing was unpacked or started",
|
||||
r.Source, r.Digest, got)
|
||||
}
|
||||
|
||||
// **Its identity is its bytes AND how it is run.** Two processes from one bundle differing
|
||||
// only in their command are different, and a record tracking the digest alone would call the
|
||||
// second one unchanged.
|
||||
want := got + " " + unitFor(r)
|
||||
at := filepath.Join(daemonRoot, r.Name)
|
||||
|
||||
// **Something it reads changed, so it must be restarted even though it is unchanged.** A
|
||||
// running process does not re-read its configuration: replace the file, find the daemon
|
||||
// already up, do nothing, and the machine keeps behaving the way it did before while every
|
||||
// check passes. The same rule a service follows, for the same reason.
|
||||
var because string
|
||||
for _, id := range r.RestartOn {
|
||||
if changed[id] {
|
||||
because = id
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
if previous.Wrote == want && because == "" {
|
||||
// Everything about it is as declared. Still asked whether it is RUNNING, because a
|
||||
// declaration that is satisfied by a record rather than by the machine is how a stopped
|
||||
// service reports success.
|
||||
if active, err := run(ctx, "systemctl", "is-active", "--quiet", r.Name+".service"); err == nil {
|
||||
_ = active
|
||||
return out, nil
|
||||
}
|
||||
if _, err := run(ctx, "systemctl", "start", r.Name+".service"); err != nil {
|
||||
return out, fmt.Errorf("%s is installed and would not start: %w", r.Name, err)
|
||||
}
|
||||
out.Action = "updated"
|
||||
out.Detail = "restarted a daemon that had stopped"
|
||||
out.wrote = want
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Replaced rather than merged: the bundle is the whole of what it runs, and files left from a
|
||||
// previous version would be loaded by a runtime that walks a directory.
|
||||
if err := os.RemoveAll(at); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := os.MkdirAll(at, 0o755); err != nil {
|
||||
return out, err
|
||||
}
|
||||
written, err := unpack(body, at)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := ownAll(at, r.User); err != nil {
|
||||
return out, err
|
||||
}
|
||||
|
||||
// **A step is run to completion, not installed.** What follows it is gated on it finishing,
|
||||
// so the machine is not asked to start something that needed a migration that did not happen.
|
||||
// Nothing is left behind to ask afterwards: the record that it ran is the digest, which is why
|
||||
// the identity above includes the command.
|
||||
if r.RunOnce {
|
||||
if _, err := run(ctx, r.Run[0], r.Run[1:]...); err != nil {
|
||||
return out, fmt.Errorf("the %s step did not complete: %w", r.Name, err)
|
||||
}
|
||||
out.Action = "created"
|
||||
if previous.Wrote != "" {
|
||||
out.Action = "updated"
|
||||
}
|
||||
out.Detail = fmt.Sprintf("%d file(s), step completed", written)
|
||||
out.wrote = want
|
||||
return out, nil
|
||||
}
|
||||
|
||||
unit := filepath.Join(unitDir, r.Name+".service")
|
||||
if err := os.WriteFile(unit, []byte(unitFor(r)), 0o644); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
||||
return out, err
|
||||
}
|
||||
// Enabled and restarted, in that order: enabled so it survives a reboot, restarted rather than
|
||||
// started because this path is also how a new version arrives and the old one is still running.
|
||||
// **Scheduled means a timer, not a service that stays up.** The unit above is written either
|
||||
// way and describes what to run; what differs is whether the machine is asked to keep it
|
||||
// running or to start it when the timer says so.
|
||||
if r.Schedule != "" {
|
||||
timer := filepath.Join(unitDir, r.Name+".timer")
|
||||
if err := os.WriteFile(timer, []byte(timerFor(r)), 0o644); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
||||
return out, err
|
||||
}
|
||||
// The timer is enabled and started; the service is neither. Enabling the service too
|
||||
// would have it run at boot as well as on its cadence, which is a second schedule nobody
|
||||
// asked for.
|
||||
if _, err := run(ctx, "systemctl", "enable", r.Name+".timer"); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if _, err := run(ctx, "systemctl", "restart", r.Name+".timer"); err != nil {
|
||||
return out, fmt.Errorf("%s was installed and its timer would not start: %w", r.Name, err)
|
||||
}
|
||||
out.Action = "updated"
|
||||
if previous.Wrote == "" {
|
||||
out.Action = "created"
|
||||
}
|
||||
out.Detail = fmt.Sprintf("%d file(s), scheduled as %s.timer", written, r.Name)
|
||||
out.wrote = want
|
||||
return out, nil
|
||||
}
|
||||
|
||||
if _, err := run(ctx, "systemctl", "enable", r.Name+".service"); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if _, err := run(ctx, "systemctl", "restart", r.Name+".service"); err != nil {
|
||||
return out, fmt.Errorf("%s was installed and would not start: %w", r.Name, err)
|
||||
}
|
||||
|
||||
out.Action = "updated"
|
||||
if previous.Wrote == "" {
|
||||
out.Action = "created"
|
||||
}
|
||||
out.Detail = fmt.Sprintf("%d file(s), running as %s.service", written, r.Name)
|
||||
if because != "" {
|
||||
out.Detail += ", restarted because " + because + " changed"
|
||||
}
|
||||
out.wrote = want
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// unitFor is the unit the mesh writes for a daemon.
|
||||
//
|
||||
// **Generated whole and never edited in place**, the same rule as every other managed file: an
|
||||
// edit survives until the next declaration and then vanishes, which is worse than not being
|
||||
// allowed at all, so the file says so.
|
||||
//
|
||||
// Deterministic — environment sorted — because this string is half the daemon's identity, and a
|
||||
// map iterated in Go's order would make every apply look like a change.
|
||||
func unitFor(r *declaration.Process) string {
|
||||
var b strings.Builder
|
||||
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
||||
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
||||
b.WriteString("[Unit]\n")
|
||||
fmt.Fprintf(&b, "Description=%s, a mesh daemon\n", r.Name)
|
||||
b.WriteString("After=network-online.target\n")
|
||||
b.WriteString("Wants=network-online.target\n\n")
|
||||
|
||||
b.WriteString("[Service]\n")
|
||||
b.WriteString("Type=simple\n")
|
||||
fmt.Fprintf(&b, "WorkingDirectory=%s\n", filepath.Join(daemonRoot, r.Name))
|
||||
for _, file := range r.EnvFile {
|
||||
fmt.Fprintf(&b, "EnvironmentFile=%s\n", file)
|
||||
}
|
||||
for _, key := range sortedKeys(r.Env) {
|
||||
fmt.Fprintf(&b, "Environment=%s=%s\n", key, r.Env[key])
|
||||
}
|
||||
if r.User != "" {
|
||||
fmt.Fprintf(&b, "User=%s\n", r.User)
|
||||
}
|
||||
fmt.Fprintf(&b, "ExecStart=%s\n", strings.Join(r.Run, " "))
|
||||
if r.Schedule != "" {
|
||||
// Started by its timer and expected to finish. Restarting it would have it run
|
||||
// continuously between fires, which is the opposite of a schedule.
|
||||
b.WriteString("Type=oneshot\n")
|
||||
b.WriteString("\n")
|
||||
return strings.Replace(b.String(), "Type=simple\n", "", 1)
|
||||
}
|
||||
// Restarted when it exits, because something that stops is not something that stays up.
|
||||
// Delayed, so a process that fails at once does not spin the machine.
|
||||
b.WriteString("Restart=always\nRestartSec=5\n\n")
|
||||
|
||||
b.WriteString("[Install]\nWantedBy=multi-user.target\n")
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// timerFor is the cadence a scheduled process runs on.
|
||||
//
|
||||
// **The mesh's cron expression, handed to the machine's own timer.** The host already parses and
|
||||
// evaluates five-field cron (ADR 0053) for a scheduled container; a machine with a supervisor can
|
||||
// be told the cadence directly rather than have the host wake up and decide.
|
||||
func timerFor(r *declaration.Process) string {
|
||||
var b strings.Builder
|
||||
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
||||
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
||||
b.WriteString("[Unit]\n")
|
||||
fmt.Fprintf(&b, "Description=%s, on a schedule the mesh set\n\n", r.Name)
|
||||
b.WriteString("[Timer]\n")
|
||||
fmt.Fprintf(&b, "OnCalendar=%s\n", calendarFor(r.Schedule))
|
||||
// A fire missed because the machine was off happens when it comes back, rather than being
|
||||
// skipped silently — which is the difference between a machine that was down and a schedule
|
||||
// that quietly stopped.
|
||||
b.WriteString("Persistent=true\n\n")
|
||||
b.WriteString("[Install]\nWantedBy=timers.target\n")
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// calendarFor turns five-field cron into what a systemd timer reads.
|
||||
//
|
||||
// minute hour day-of-month month day-of-week -> DayOfWeek Year-Month-Day Hour:Minute:Second
|
||||
func calendarFor(cron string) string {
|
||||
fields := strings.Fields(cron)
|
||||
if len(fields) != 5 {
|
||||
// Refused at validation, so this is unreachable — and returning something that would fire
|
||||
// constantly is worse than returning something that never does.
|
||||
return "*-*-* 00:00:00"
|
||||
}
|
||||
minute, hour, dom, month, dow := fields[0], fields[1], fields[2], fields[3], fields[4]
|
||||
day := dow
|
||||
if dow == "*" {
|
||||
day = ""
|
||||
} else {
|
||||
day += " "
|
||||
}
|
||||
return fmt.Sprintf("%s*-%s-%s %s:%s:00", day, month, dom, hour, minute)
|
||||
}
|
||||
Reference in New Issue
Block a user