The first cut of this added a `daemon` for the long-running case alone. That would have meant a new vocabulary entry for each of the others — a scheduled task, a run-once migration, a health check — when they are one thing run at different cadences. That is a field, not four entries in a vocabulary where every entry widens what a compromised control plane can express. So it mirrors a container exactly, because it IS a container's twin: the same intent, hosted by the machine's own supervisor instead of a runtime. Stays up, runs once, or runs on a schedule. Tools, hooks and event consumers are not further modes. They are loaded by a tool host, which is itself a process that stays up — so the generic case already covers them, which is the test of whether it is generic. A scheduled process gets a timer and a unit that finishes; a long-running one gets a unit that is restarted when it exits. Getting that wrong either way is a second copy running continuously between fires, or a schedule that never fires. The modes are exclusive and validation says so near the author: something that runs once does not run on a schedule, and something not running between fires cannot be restarted when a file changes. A missed fire happens when the machine comes back rather than being skipped, which is the difference between a machine that was down and a schedule that quietly stopped. Claude-Session: https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx
267 lines
10 KiB
Go
267 lines
10 KiB
Go
package apply
|
|
|
|
import (
|
|
"context"
|
|
"crypto/sha256"
|
|
"encoding/hex"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
|
|
"github.com/novox/mesh-host/internal/declaration"
|
|
"github.com/novox/mesh-host/internal/store"
|
|
)
|
|
|
|
// Running the mesh's own code, without the module choosing how.
|
|
//
|
|
// **A daemon is an intent and this is one answer to it.** A module says what to run and the
|
|
// machine's own supervisor is how — which means the module is not writing a unit file, and not
|
|
// choosing between a container and a service before it can declare anything
|
|
// (novox/hq 03-DESIGN/01-to-be/18-building-a-module.md).
|
|
//
|
|
// What this does, in order: fetch the bundle, refuse it unless it hashes to what was declared,
|
|
// unpack it where the mesh keeps such things, write the unit, and put it in the state asked for.
|
|
// The unit is the mesh's — an operator editing it loses the edit at the next declaration, which is
|
|
// the same rule every managed file on a machine follows (ADR 0011).
|
|
|
|
// daemonRoot is where unpacked daemons live.
|
|
//
|
|
// Under the mesh's own directory rather than somewhere a distribution owns: these are files the
|
|
// mesh puts there and replaces, and putting them where a package manager also writes is how two
|
|
// owners end up disagreeing about one path.
|
|
const daemonRoot = "/var/lib/mesh/daemons"
|
|
|
|
// unitDir is where the mesh writes the units it owns.
|
|
const unitDir = "/etc/systemd/system"
|
|
|
|
func applyProcess(ctx context.Context, r *declaration.Process, run Runner,
|
|
changed map[string]bool, previous store.Applied) (Outcome, error) {
|
|
out := begin(r)
|
|
out.Action = "unchanged"
|
|
|
|
body, err := fetch(ctx, r.Source)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
sum := sha256.Sum256(body)
|
|
got := "sha256:" + hex.EncodeToString(sum[:])
|
|
if got != r.Digest {
|
|
// Refused before anything is written or started. What is at that address is not what was
|
|
// declared, and running it would be running something nobody reviewed.
|
|
return out, fmt.Errorf(
|
|
"%s was declared as %s and what arrived is %s; nothing was unpacked or started",
|
|
r.Source, r.Digest, got)
|
|
}
|
|
|
|
// **Its identity is its bytes AND how it is run.** Two processes from one bundle differing
|
|
// only in their command are different, and a record tracking the digest alone would call the
|
|
// second one unchanged.
|
|
want := got + " " + unitFor(r)
|
|
at := filepath.Join(daemonRoot, r.Name)
|
|
|
|
// **Something it reads changed, so it must be restarted even though it is unchanged.** A
|
|
// running process does not re-read its configuration: replace the file, find the daemon
|
|
// already up, do nothing, and the machine keeps behaving the way it did before while every
|
|
// check passes. The same rule a service follows, for the same reason.
|
|
var because string
|
|
for _, id := range r.RestartOn {
|
|
if changed[id] {
|
|
because = id
|
|
break
|
|
}
|
|
}
|
|
|
|
if previous.Wrote == want && because == "" {
|
|
// Everything about it is as declared. Still asked whether it is RUNNING, because a
|
|
// declaration that is satisfied by a record rather than by the machine is how a stopped
|
|
// service reports success.
|
|
if active, err := run(ctx, "systemctl", "is-active", "--quiet", r.Name+".service"); err == nil {
|
|
_ = active
|
|
return out, nil
|
|
}
|
|
if _, err := run(ctx, "systemctl", "start", r.Name+".service"); err != nil {
|
|
return out, fmt.Errorf("%s is installed and would not start: %w", r.Name, err)
|
|
}
|
|
out.Action = "updated"
|
|
out.Detail = "restarted a daemon that had stopped"
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
// Replaced rather than merged: the bundle is the whole of what it runs, and files left from a
|
|
// previous version would be loaded by a runtime that walks a directory.
|
|
if err := os.RemoveAll(at); err != nil {
|
|
return out, err
|
|
}
|
|
if err := os.MkdirAll(at, 0o755); err != nil {
|
|
return out, err
|
|
}
|
|
written, err := unpack(body, at)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if err := ownAll(at, r.User); err != nil {
|
|
return out, err
|
|
}
|
|
|
|
// **A step is run to completion, not installed.** What follows it is gated on it finishing,
|
|
// so the machine is not asked to start something that needed a migration that did not happen.
|
|
// Nothing is left behind to ask afterwards: the record that it ran is the digest, which is why
|
|
// the identity above includes the command.
|
|
if r.RunOnce {
|
|
if _, err := run(ctx, r.Run[0], r.Run[1:]...); err != nil {
|
|
return out, fmt.Errorf("the %s step did not complete: %w", r.Name, err)
|
|
}
|
|
out.Action = "created"
|
|
if previous.Wrote != "" {
|
|
out.Action = "updated"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), step completed", written)
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
unit := filepath.Join(unitDir, r.Name+".service")
|
|
if err := os.WriteFile(unit, []byte(unitFor(r)), 0o644); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
|
return out, err
|
|
}
|
|
// Enabled and restarted, in that order: enabled so it survives a reboot, restarted rather than
|
|
// started because this path is also how a new version arrives and the old one is still running.
|
|
// **Scheduled means a timer, not a service that stays up.** The unit above is written either
|
|
// way and describes what to run; what differs is whether the machine is asked to keep it
|
|
// running or to start it when the timer says so.
|
|
if r.Schedule != "" {
|
|
timer := filepath.Join(unitDir, r.Name+".timer")
|
|
if err := os.WriteFile(timer, []byte(timerFor(r)), 0o644); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
|
return out, err
|
|
}
|
|
// The timer is enabled and started; the service is neither. Enabling the service too
|
|
// would have it run at boot as well as on its cadence, which is a second schedule nobody
|
|
// asked for.
|
|
if _, err := run(ctx, "systemctl", "enable", r.Name+".timer"); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "restart", r.Name+".timer"); err != nil {
|
|
return out, fmt.Errorf("%s was installed and its timer would not start: %w", r.Name, err)
|
|
}
|
|
out.Action = "updated"
|
|
if previous.Wrote == "" {
|
|
out.Action = "created"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), scheduled as %s.timer", written, r.Name)
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
if _, err := run(ctx, "systemctl", "enable", r.Name+".service"); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "restart", r.Name+".service"); err != nil {
|
|
return out, fmt.Errorf("%s was installed and would not start: %w", r.Name, err)
|
|
}
|
|
|
|
out.Action = "updated"
|
|
if previous.Wrote == "" {
|
|
out.Action = "created"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), running as %s.service", written, r.Name)
|
|
if because != "" {
|
|
out.Detail += ", restarted because " + because + " changed"
|
|
}
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
// unitFor is the unit the mesh writes for a daemon.
|
|
//
|
|
// **Generated whole and never edited in place**, the same rule as every other managed file: an
|
|
// edit survives until the next declaration and then vanishes, which is worse than not being
|
|
// allowed at all, so the file says so.
|
|
//
|
|
// Deterministic — environment sorted — because this string is half the daemon's identity, and a
|
|
// map iterated in Go's order would make every apply look like a change.
|
|
func unitFor(r *declaration.Process) string {
|
|
var b strings.Builder
|
|
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
|
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
|
b.WriteString("[Unit]\n")
|
|
fmt.Fprintf(&b, "Description=%s, a mesh daemon\n", r.Name)
|
|
b.WriteString("After=network-online.target\n")
|
|
b.WriteString("Wants=network-online.target\n\n")
|
|
|
|
b.WriteString("[Service]\n")
|
|
b.WriteString("Type=simple\n")
|
|
fmt.Fprintf(&b, "WorkingDirectory=%s\n", filepath.Join(daemonRoot, r.Name))
|
|
for _, file := range r.EnvFile {
|
|
fmt.Fprintf(&b, "EnvironmentFile=%s\n", file)
|
|
}
|
|
for _, key := range sortedKeys(r.Env) {
|
|
fmt.Fprintf(&b, "Environment=%s=%s\n", key, r.Env[key])
|
|
}
|
|
if r.User != "" {
|
|
fmt.Fprintf(&b, "User=%s\n", r.User)
|
|
}
|
|
fmt.Fprintf(&b, "ExecStart=%s\n", strings.Join(r.Run, " "))
|
|
if r.Schedule != "" {
|
|
// Started by its timer and expected to finish. Restarting it would have it run
|
|
// continuously between fires, which is the opposite of a schedule.
|
|
b.WriteString("Type=oneshot\n")
|
|
b.WriteString("\n")
|
|
return strings.Replace(b.String(), "Type=simple\n", "", 1)
|
|
}
|
|
// Restarted when it exits, because something that stops is not something that stays up.
|
|
// Delayed, so a process that fails at once does not spin the machine.
|
|
b.WriteString("Restart=always\nRestartSec=5\n\n")
|
|
|
|
b.WriteString("[Install]\nWantedBy=multi-user.target\n")
|
|
return b.String()
|
|
}
|
|
|
|
// timerFor is the cadence a scheduled process runs on.
|
|
//
|
|
// **The mesh's cron expression, handed to the machine's own timer.** The host already parses and
|
|
// evaluates five-field cron (ADR 0053) for a scheduled container; a machine with a supervisor can
|
|
// be told the cadence directly rather than have the host wake up and decide.
|
|
func timerFor(r *declaration.Process) string {
|
|
var b strings.Builder
|
|
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
|
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
|
b.WriteString("[Unit]\n")
|
|
fmt.Fprintf(&b, "Description=%s, on a schedule the mesh set\n\n", r.Name)
|
|
b.WriteString("[Timer]\n")
|
|
fmt.Fprintf(&b, "OnCalendar=%s\n", calendarFor(r.Schedule))
|
|
// A fire missed because the machine was off happens when it comes back, rather than being
|
|
// skipped silently — which is the difference between a machine that was down and a schedule
|
|
// that quietly stopped.
|
|
b.WriteString("Persistent=true\n\n")
|
|
b.WriteString("[Install]\nWantedBy=timers.target\n")
|
|
return b.String()
|
|
}
|
|
|
|
// calendarFor turns five-field cron into what a systemd timer reads.
|
|
//
|
|
// minute hour day-of-month month day-of-week -> DayOfWeek Year-Month-Day Hour:Minute:Second
|
|
func calendarFor(cron string) string {
|
|
fields := strings.Fields(cron)
|
|
if len(fields) != 5 {
|
|
// Refused at validation, so this is unreachable — and returning something that would fire
|
|
// constantly is worse than returning something that never does.
|
|
return "*-*-* 00:00:00"
|
|
}
|
|
minute, hour, dom, month, dow := fields[0], fields[1], fields[2], fields[3], fields[4]
|
|
day := dow
|
|
if dow == "*" {
|
|
day = ""
|
|
} else {
|
|
day += " "
|
|
}
|
|
return fmt.Sprintf("%s*-%s-%s %s:%s:00", day, month, dom, hour, minute)
|
|
}
|