A service undeclared used to be stopped: unassigning the private network stopped the container runtime, unassigning sshd would stop ssh, an uplink module would take the machine offline. The host now records the unit's state when it first applies it and restores that on undeclare — found running stays running; started by the mesh (the converge filter) is stopped again; nothing is started on the way out; a pre-existing record leaves the unit alone. An undeclared process had no removal at all and failed every apply on its node; its unit, timer and bundle are now removed.
343 lines
14 KiB
Go
343 lines
14 KiB
Go
package apply
|
|
|
|
import (
|
|
"context"
|
|
"crypto/sha256"
|
|
"encoding/hex"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
|
|
"github.com/novox/mesh-host/internal/declaration"
|
|
"github.com/novox/mesh-host/internal/store"
|
|
)
|
|
|
|
// Running the mesh's own code, without the module choosing how.
|
|
//
|
|
// **A daemon is an intent and this is one answer to it.** A module says what to run and the
|
|
// machine's own supervisor is how — which means the module is not writing a unit file, and not
|
|
// choosing between a container and a service before it can declare anything
|
|
// (novox/hq 03-DESIGN/01-to-be/18-building-a-module.md).
|
|
//
|
|
// What this does, in order: fetch the bundle, refuse it unless it hashes to what was declared,
|
|
// unpack it where the mesh keeps such things, write the unit, and put it in the state asked for.
|
|
// The unit is the mesh's — an operator editing it loses the edit at the next declaration, which is
|
|
// the same rule every managed file on a machine follows (ADR 0011).
|
|
|
|
// daemonRoot is where unpacked daemons live.
|
|
//
|
|
// Under the mesh's own directory rather than somewhere a distribution owns: these are files the
|
|
// mesh puts there and replaces, and putting them where a package manager also writes is how two
|
|
// owners end up disagreeing about one path.
|
|
var daemonRoot = "/var/lib/mesh/daemons"
|
|
|
|
// unitDir is where the mesh writes the units it owns. A variable only so a test can point it at a
|
|
// directory of its own.
|
|
var unitDir = "/etc/systemd/system"
|
|
|
|
func applyProcess(ctx context.Context, r *declaration.Process, run Runner,
|
|
changed map[string]bool, previous store.Applied) (Outcome, error) {
|
|
out := begin(r)
|
|
out.Action = "unchanged"
|
|
|
|
body, err := fetch(ctx, r.Source)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
sum := sha256.Sum256(body)
|
|
got := "sha256:" + hex.EncodeToString(sum[:])
|
|
if got != r.Digest {
|
|
// Refused before anything is written or started. What is at that address is not what was
|
|
// declared, and running it would be running something nobody reviewed.
|
|
return out, fmt.Errorf(
|
|
"%s was declared as %s and what arrived is %s; nothing was unpacked or started",
|
|
r.Source, r.Digest, got)
|
|
}
|
|
|
|
// **Its identity is its bytes AND how it is run.** Two processes from one bundle differing
|
|
// only in their command are different, and a record tracking the digest alone would call the
|
|
// second one unchanged.
|
|
want := got + " " + unitFor(r)
|
|
at := filepath.Join(daemonRoot, r.Name)
|
|
|
|
// **Something it reads changed, so it must be restarted even though it is unchanged.** A
|
|
// running process does not re-read its configuration: replace the file, find the daemon
|
|
// already up, do nothing, and the machine keeps behaving the way it did before while every
|
|
// check passes. The same rule a service follows, for the same reason.
|
|
var because string
|
|
for _, id := range r.RestartOn {
|
|
if changed[id] {
|
|
because = id
|
|
break
|
|
}
|
|
}
|
|
|
|
if previous.Wrote == want && because == "" {
|
|
// Everything about it is as declared. Still asked whether it is RUNNING, because a
|
|
// declaration that is satisfied by a record rather than by the machine is how a stopped
|
|
// service reports success.
|
|
if active, err := run(ctx, "systemctl", "is-active", "--quiet", r.Name+".service"); err == nil {
|
|
_ = active
|
|
return out, nil
|
|
}
|
|
if _, err := run(ctx, "systemctl", "start", r.Name+".service"); err != nil {
|
|
return out, fmt.Errorf("%s is installed and would not start: %w", r.Name, err)
|
|
}
|
|
out.Action = "updated"
|
|
out.Detail = "restarted a daemon that had stopped"
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
// Replaced rather than merged: the bundle is the whole of what it runs, and files left from a
|
|
// previous version would be loaded by a runtime that walks a directory.
|
|
if err := os.RemoveAll(at); err != nil {
|
|
return out, err
|
|
}
|
|
if err := os.MkdirAll(at, 0o755); err != nil {
|
|
return out, err
|
|
}
|
|
written, err := unpack(body, at)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if err := ownAll(at, r.User); err != nil {
|
|
return out, err
|
|
}
|
|
|
|
// **A step is run to completion, not installed.** What follows it is gated on it finishing,
|
|
// so the machine is not asked to start something that needed a migration that did not happen.
|
|
// Nothing is left behind to ask afterwards: the record that it ran is the digest, which is why
|
|
// the identity above includes the command.
|
|
if r.RunOnce {
|
|
if _, err := run(ctx, r.Run[0], r.Run[1:]...); err != nil {
|
|
return out, fmt.Errorf("the %s step did not complete: %w", r.Name, err)
|
|
}
|
|
out.Action = "created"
|
|
if previous.Wrote != "" {
|
|
out.Action = "updated"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), step completed", written)
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
unit := filepath.Join(unitDir, r.Name+".service")
|
|
if err := os.WriteFile(unit, []byte(unitFor(r)), 0o644); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
|
return out, err
|
|
}
|
|
// Enabled and restarted, in that order: enabled so it survives a reboot, restarted rather than
|
|
// started because this path is also how a new version arrives and the old one is still running.
|
|
// **Scheduled means a timer, not a service that stays up.** The unit above is written either
|
|
// way and describes what to run; what differs is whether the machine is asked to keep it
|
|
// running or to start it when the timer says so.
|
|
if r.Schedule != "" {
|
|
timer := filepath.Join(unitDir, r.Name+".timer")
|
|
if err := os.WriteFile(timer, []byte(timerFor(r)), 0o644); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
|
return out, err
|
|
}
|
|
// The timer is enabled and started; the service is neither. Enabling the service too
|
|
// would have it run at boot as well as on its cadence, which is a second schedule nobody
|
|
// asked for.
|
|
if _, err := run(ctx, "systemctl", "enable", r.Name+".timer"); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "restart", r.Name+".timer"); err != nil {
|
|
return out, fmt.Errorf("%s was installed and its timer would not start: %w", r.Name, err)
|
|
}
|
|
out.Action = "updated"
|
|
if previous.Wrote == "" {
|
|
out.Action = "created"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), scheduled as %s.timer", written, r.Name)
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
if _, err := run(ctx, "systemctl", "enable", r.Name+".service"); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "restart", r.Name+".service"); err != nil {
|
|
return out, fmt.Errorf("%s was installed and would not start: %w", r.Name, err)
|
|
}
|
|
|
|
out.Action = "updated"
|
|
if previous.Wrote == "" {
|
|
out.Action = "created"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), running as %s.service", written, r.Name)
|
|
if because != "" {
|
|
out.Detail += ", restarted because " + because + " changed"
|
|
}
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
// unitFor is the unit the mesh writes for a daemon.
|
|
//
|
|
// **Generated whole and never edited in place**, the same rule as every other managed file: an
|
|
// edit survives until the next declaration and then vanishes, which is worse than not being
|
|
// allowed at all, so the file says so.
|
|
//
|
|
// Deterministic — environment sorted — because this string is half the daemon's identity, and a
|
|
// map iterated in Go's order would make every apply look like a change.
|
|
func unitFor(r *declaration.Process) string {
|
|
var b strings.Builder
|
|
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
|
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
|
b.WriteString("[Unit]\n")
|
|
fmt.Fprintf(&b, "Description=%s, a mesh daemon\n", r.Name)
|
|
b.WriteString("After=network-online.target\n")
|
|
b.WriteString("Wants=network-online.target\n\n")
|
|
|
|
b.WriteString("[Service]\n")
|
|
b.WriteString("Type=simple\n")
|
|
fmt.Fprintf(&b, "WorkingDirectory=%s\n", filepath.Join(daemonRoot, r.Name))
|
|
for _, file := range r.EnvFile {
|
|
fmt.Fprintf(&b, "EnvironmentFile=%s\n", file)
|
|
}
|
|
for _, key := range sortedKeys(r.Env) {
|
|
fmt.Fprintf(&b, "Environment=%s\n", unitValue(key, r.Env[key]))
|
|
}
|
|
if r.User != "" {
|
|
fmt.Fprintf(&b, "User=%s\n", r.User)
|
|
}
|
|
fmt.Fprintf(&b, "ExecStart=%s\n", strings.Join(r.Run, " "))
|
|
if r.Schedule != "" {
|
|
// Started by its timer and expected to finish. Restarting it would have it run
|
|
// continuously between fires, which is the opposite of a schedule.
|
|
b.WriteString("Type=oneshot\n")
|
|
b.WriteString("\n")
|
|
return strings.Replace(b.String(), "Type=simple\n", "", 1)
|
|
}
|
|
// Restarted when it exits, because something that stops is not something that stays up.
|
|
// Delayed, so a process that fails at once does not spin the machine.
|
|
b.WriteString("Restart=always\nRestartSec=5\n\n")
|
|
|
|
b.WriteString("[Install]\nWantedBy=multi-user.target\n")
|
|
return b.String()
|
|
}
|
|
|
|
// timerFor is the cadence a scheduled process runs on.
|
|
//
|
|
// **The mesh's cron expression, handed to the machine's own timer.** The host already parses and
|
|
// evaluates five-field cron (ADR 0053) for a scheduled container; a machine with a supervisor can
|
|
// be told the cadence directly rather than have the host wake up and decide.
|
|
func timerFor(r *declaration.Process) string {
|
|
var b strings.Builder
|
|
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
|
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
|
b.WriteString("[Unit]\n")
|
|
fmt.Fprintf(&b, "Description=%s, on a schedule the mesh set\n\n", r.Name)
|
|
b.WriteString("[Timer]\n")
|
|
fmt.Fprintf(&b, "OnCalendar=%s\n", calendarFor(r.Schedule))
|
|
// A fire missed because the machine was off happens when it comes back, rather than being
|
|
// skipped silently — which is the difference between a machine that was down and a schedule
|
|
// that quietly stopped.
|
|
b.WriteString("Persistent=true\n\n")
|
|
b.WriteString("[Install]\nWantedBy=timers.target\n")
|
|
return b.String()
|
|
}
|
|
|
|
// calendarFor turns five-field cron into what a systemd timer reads.
|
|
//
|
|
// minute hour day-of-month month day-of-week -> DayOfWeek Year-Month-Day Hour:Minute:Second
|
|
func calendarFor(cron string) string {
|
|
fields := strings.Fields(cron)
|
|
if len(fields) != 5 {
|
|
// Refused at validation, so this is unreachable — and returning something that would fire
|
|
// constantly is worse than returning something that never does.
|
|
return "*-*-* 00:00:00"
|
|
}
|
|
minute, hour, dom, month, dow := fields[0], fields[1], fields[2], fields[3], fields[4]
|
|
day := dow
|
|
if dow == "*" {
|
|
day = ""
|
|
} else {
|
|
day += " "
|
|
}
|
|
return fmt.Sprintf("%s*-%s-%s %s:%s:00", day, month, dom, hour, minute)
|
|
}
|
|
|
|
// unitValue renders one environment assignment so a unit file means what the declaration said.
|
|
//
|
|
// **Three things a unit file does to a value that nothing else does**, and a module's environment
|
|
// routinely contains all three — a generated password is arbitrary bytes.
|
|
//
|
|
// - `%` begins a specifier. `%H` is the hostname, `%i` the instance. A password containing one
|
|
// is silently replaced by something else, and the failure is an authentication error nobody
|
|
// can explain by looking at the declaration.
|
|
// - whitespace separates assignments. `Environment=K=a b` sets K to "a" and then tries to read
|
|
// "b" as another assignment.
|
|
// - a newline ends the line. What follows it is read as a unit DIRECTIVE, so a value carrying
|
|
// one could write ExecStart= and have the machine run something nobody declared.
|
|
//
|
|
// Quoted, with quotes and backslashes escaped and percent doubled. A container needs none of this
|
|
// because `--env` is passed through literally, which is why this had to be found here rather than
|
|
// noticed in both.
|
|
func unitValue(key, value string) string {
|
|
escaped := strings.NewReplacer(
|
|
`\`, `\\`,
|
|
`"`, `\"`,
|
|
"%", "%%",
|
|
).Replace(value)
|
|
return `"` + key + "=" + escaped + `"`
|
|
}
|
|
|
|
// removeProcess takes away a process the mesh no longer declares: its timer and unit stopped and
|
|
// disabled, their files removed, the supervisor told, and the unpacked bundle deleted.
|
|
//
|
|
// **All of it is the host's**, which is why all of it goes (novox/hq ADR 0118). Before this there
|
|
// was no way to remove a process at all, and one left undeclared failed every apply on its node
|
|
// until someone edited the host's state by hand — unassigning any module that ran its own code
|
|
// stranded the machine.
|
|
//
|
|
// Idempotent, like every removal: a unit already gone is not an error, and a record whose files
|
|
// have all vanished is forgotten rather than reported as removed.
|
|
func removeProcess(ctx context.Context, a store.Applied, run Runner) (string, string, error) {
|
|
name := a.Target
|
|
if name == "" || strings.ContainsAny(name, "/ \t") {
|
|
// The name is a unit name and a directory under the mesh's own. One that could climb out
|
|
// of either is refused rather than acted on, whatever wrote it into the record.
|
|
return "", "", fmt.Errorf("a process recorded under %q cannot be removed by name", name)
|
|
}
|
|
found := false
|
|
for _, unit := range []string{name + ".timer", name + ".service"} {
|
|
path := filepath.Join(unitDir, unit)
|
|
if _, err := os.Stat(path); err != nil {
|
|
continue
|
|
}
|
|
found = true
|
|
// The timer first, so a scheduled run cannot start the service between the two.
|
|
if _, err := run(ctx, "systemctl", "disable", "--now", unit); err != nil {
|
|
return "", "", fmt.Errorf("stopping %s: %w", unit, err)
|
|
}
|
|
if err := os.Remove(path); err != nil && !os.IsNotExist(err) {
|
|
return "", "", err
|
|
}
|
|
}
|
|
if found {
|
|
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
|
return "", "", err
|
|
}
|
|
}
|
|
bundle := filepath.Join(daemonRoot, name)
|
|
if _, err := os.Stat(bundle); err == nil {
|
|
found = true
|
|
if err := os.RemoveAll(bundle); err != nil {
|
|
return "", "", err
|
|
}
|
|
}
|
|
if !found {
|
|
return "forgotten", "no longer there", nil
|
|
}
|
|
return "removed", "stopped; its unit and its bundle removed — the mesh's own code", nil
|
|
}
|