Found by being asked whether processes and containers handle environment the
same way. They do not, and the difference is not cosmetic.
Docker passes --env through literally. A unit file reads three things out of a
value that nothing else does, and a module's environment routinely contains all
three because a generated password is arbitrary bytes:
- % begins a specifier. %H is the hostname. A password containing one is
silently replaced, and it fails later as an authentication error nobody can
explain by reading the declaration.
- whitespace separates assignments. Unquoted, K=a b sets K to "a" and reads
"b" as another assignment.
- a newline ends the line, and what follows is read as a unit DIRECTIVE.
The first two are escaped: quoted, with quotes and backslashes escaped and
percent doubled. The third cannot be — a unit's environment has no way to carry
a line break — so it is refused in validation, near whoever wrote it. Without
that, an environment value could write ExecStart= and have the machine run
something nobody declared.
Ordinary awkward values stay accepted, because refusing those too would leave a
module unable to hold a generated password.
Claude-Session: https://claude.ai/code/session_01D6qtiYU3P9jk3pnAXyAFyx
292 lines
12 KiB
Go
292 lines
12 KiB
Go
package apply
|
|
|
|
import (
|
|
"context"
|
|
"crypto/sha256"
|
|
"encoding/hex"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
|
|
"github.com/novox/mesh-host/internal/declaration"
|
|
"github.com/novox/mesh-host/internal/store"
|
|
)
|
|
|
|
// Running the mesh's own code, without the module choosing how.
|
|
//
|
|
// **A daemon is an intent and this is one answer to it.** A module says what to run and the
|
|
// machine's own supervisor is how — which means the module is not writing a unit file, and not
|
|
// choosing between a container and a service before it can declare anything
|
|
// (novox/hq 03-DESIGN/01-to-be/18-building-a-module.md).
|
|
//
|
|
// What this does, in order: fetch the bundle, refuse it unless it hashes to what was declared,
|
|
// unpack it where the mesh keeps such things, write the unit, and put it in the state asked for.
|
|
// The unit is the mesh's — an operator editing it loses the edit at the next declaration, which is
|
|
// the same rule every managed file on a machine follows (ADR 0011).
|
|
|
|
// daemonRoot is where unpacked daemons live.
|
|
//
|
|
// Under the mesh's own directory rather than somewhere a distribution owns: these are files the
|
|
// mesh puts there and replaces, and putting them where a package manager also writes is how two
|
|
// owners end up disagreeing about one path.
|
|
const daemonRoot = "/var/lib/mesh/daemons"
|
|
|
|
// unitDir is where the mesh writes the units it owns.
|
|
const unitDir = "/etc/systemd/system"
|
|
|
|
func applyProcess(ctx context.Context, r *declaration.Process, run Runner,
|
|
changed map[string]bool, previous store.Applied) (Outcome, error) {
|
|
out := begin(r)
|
|
out.Action = "unchanged"
|
|
|
|
body, err := fetch(ctx, r.Source)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
sum := sha256.Sum256(body)
|
|
got := "sha256:" + hex.EncodeToString(sum[:])
|
|
if got != r.Digest {
|
|
// Refused before anything is written or started. What is at that address is not what was
|
|
// declared, and running it would be running something nobody reviewed.
|
|
return out, fmt.Errorf(
|
|
"%s was declared as %s and what arrived is %s; nothing was unpacked or started",
|
|
r.Source, r.Digest, got)
|
|
}
|
|
|
|
// **Its identity is its bytes AND how it is run.** Two processes from one bundle differing
|
|
// only in their command are different, and a record tracking the digest alone would call the
|
|
// second one unchanged.
|
|
want := got + " " + unitFor(r)
|
|
at := filepath.Join(daemonRoot, r.Name)
|
|
|
|
// **Something it reads changed, so it must be restarted even though it is unchanged.** A
|
|
// running process does not re-read its configuration: replace the file, find the daemon
|
|
// already up, do nothing, and the machine keeps behaving the way it did before while every
|
|
// check passes. The same rule a service follows, for the same reason.
|
|
var because string
|
|
for _, id := range r.RestartOn {
|
|
if changed[id] {
|
|
because = id
|
|
break
|
|
}
|
|
}
|
|
|
|
if previous.Wrote == want && because == "" {
|
|
// Everything about it is as declared. Still asked whether it is RUNNING, because a
|
|
// declaration that is satisfied by a record rather than by the machine is how a stopped
|
|
// service reports success.
|
|
if active, err := run(ctx, "systemctl", "is-active", "--quiet", r.Name+".service"); err == nil {
|
|
_ = active
|
|
return out, nil
|
|
}
|
|
if _, err := run(ctx, "systemctl", "start", r.Name+".service"); err != nil {
|
|
return out, fmt.Errorf("%s is installed and would not start: %w", r.Name, err)
|
|
}
|
|
out.Action = "updated"
|
|
out.Detail = "restarted a daemon that had stopped"
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
// Replaced rather than merged: the bundle is the whole of what it runs, and files left from a
|
|
// previous version would be loaded by a runtime that walks a directory.
|
|
if err := os.RemoveAll(at); err != nil {
|
|
return out, err
|
|
}
|
|
if err := os.MkdirAll(at, 0o755); err != nil {
|
|
return out, err
|
|
}
|
|
written, err := unpack(body, at)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if err := ownAll(at, r.User); err != nil {
|
|
return out, err
|
|
}
|
|
|
|
// **A step is run to completion, not installed.** What follows it is gated on it finishing,
|
|
// so the machine is not asked to start something that needed a migration that did not happen.
|
|
// Nothing is left behind to ask afterwards: the record that it ran is the digest, which is why
|
|
// the identity above includes the command.
|
|
if r.RunOnce {
|
|
if _, err := run(ctx, r.Run[0], r.Run[1:]...); err != nil {
|
|
return out, fmt.Errorf("the %s step did not complete: %w", r.Name, err)
|
|
}
|
|
out.Action = "created"
|
|
if previous.Wrote != "" {
|
|
out.Action = "updated"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), step completed", written)
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
unit := filepath.Join(unitDir, r.Name+".service")
|
|
if err := os.WriteFile(unit, []byte(unitFor(r)), 0o644); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
|
return out, err
|
|
}
|
|
// Enabled and restarted, in that order: enabled so it survives a reboot, restarted rather than
|
|
// started because this path is also how a new version arrives and the old one is still running.
|
|
// **Scheduled means a timer, not a service that stays up.** The unit above is written either
|
|
// way and describes what to run; what differs is whether the machine is asked to keep it
|
|
// running or to start it when the timer says so.
|
|
if r.Schedule != "" {
|
|
timer := filepath.Join(unitDir, r.Name+".timer")
|
|
if err := os.WriteFile(timer, []byte(timerFor(r)), 0o644); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "daemon-reload"); err != nil {
|
|
return out, err
|
|
}
|
|
// The timer is enabled and started; the service is neither. Enabling the service too
|
|
// would have it run at boot as well as on its cadence, which is a second schedule nobody
|
|
// asked for.
|
|
if _, err := run(ctx, "systemctl", "enable", r.Name+".timer"); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "restart", r.Name+".timer"); err != nil {
|
|
return out, fmt.Errorf("%s was installed and its timer would not start: %w", r.Name, err)
|
|
}
|
|
out.Action = "updated"
|
|
if previous.Wrote == "" {
|
|
out.Action = "created"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), scheduled as %s.timer", written, r.Name)
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
if _, err := run(ctx, "systemctl", "enable", r.Name+".service"); err != nil {
|
|
return out, err
|
|
}
|
|
if _, err := run(ctx, "systemctl", "restart", r.Name+".service"); err != nil {
|
|
return out, fmt.Errorf("%s was installed and would not start: %w", r.Name, err)
|
|
}
|
|
|
|
out.Action = "updated"
|
|
if previous.Wrote == "" {
|
|
out.Action = "created"
|
|
}
|
|
out.Detail = fmt.Sprintf("%d file(s), running as %s.service", written, r.Name)
|
|
if because != "" {
|
|
out.Detail += ", restarted because " + because + " changed"
|
|
}
|
|
out.wrote = want
|
|
return out, nil
|
|
}
|
|
|
|
// unitFor is the unit the mesh writes for a daemon.
|
|
//
|
|
// **Generated whole and never edited in place**, the same rule as every other managed file: an
|
|
// edit survives until the next declaration and then vanishes, which is worse than not being
|
|
// allowed at all, so the file says so.
|
|
//
|
|
// Deterministic — environment sorted — because this string is half the daemon's identity, and a
|
|
// map iterated in Go's order would make every apply look like a change.
|
|
func unitFor(r *declaration.Process) string {
|
|
var b strings.Builder
|
|
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
|
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
|
b.WriteString("[Unit]\n")
|
|
fmt.Fprintf(&b, "Description=%s, a mesh daemon\n", r.Name)
|
|
b.WriteString("After=network-online.target\n")
|
|
b.WriteString("Wants=network-online.target\n\n")
|
|
|
|
b.WriteString("[Service]\n")
|
|
b.WriteString("Type=simple\n")
|
|
fmt.Fprintf(&b, "WorkingDirectory=%s\n", filepath.Join(daemonRoot, r.Name))
|
|
for _, file := range r.EnvFile {
|
|
fmt.Fprintf(&b, "EnvironmentFile=%s\n", file)
|
|
}
|
|
for _, key := range sortedKeys(r.Env) {
|
|
fmt.Fprintf(&b, "Environment=%s\n", unitValue(key, r.Env[key]))
|
|
}
|
|
if r.User != "" {
|
|
fmt.Fprintf(&b, "User=%s\n", r.User)
|
|
}
|
|
fmt.Fprintf(&b, "ExecStart=%s\n", strings.Join(r.Run, " "))
|
|
if r.Schedule != "" {
|
|
// Started by its timer and expected to finish. Restarting it would have it run
|
|
// continuously between fires, which is the opposite of a schedule.
|
|
b.WriteString("Type=oneshot\n")
|
|
b.WriteString("\n")
|
|
return strings.Replace(b.String(), "Type=simple\n", "", 1)
|
|
}
|
|
// Restarted when it exits, because something that stops is not something that stays up.
|
|
// Delayed, so a process that fails at once does not spin the machine.
|
|
b.WriteString("Restart=always\nRestartSec=5\n\n")
|
|
|
|
b.WriteString("[Install]\nWantedBy=multi-user.target\n")
|
|
return b.String()
|
|
}
|
|
|
|
// timerFor is the cadence a scheduled process runs on.
|
|
//
|
|
// **The mesh's cron expression, handed to the machine's own timer.** The host already parses and
|
|
// evaluates five-field cron (ADR 0053) for a scheduled container; a machine with a supervisor can
|
|
// be told the cadence directly rather than have the host wake up and decide.
|
|
func timerFor(r *declaration.Process) string {
|
|
var b strings.Builder
|
|
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
|
b.WriteString("# declaration changes, and an edit would survive until then and vanish.\n")
|
|
b.WriteString("[Unit]\n")
|
|
fmt.Fprintf(&b, "Description=%s, on a schedule the mesh set\n\n", r.Name)
|
|
b.WriteString("[Timer]\n")
|
|
fmt.Fprintf(&b, "OnCalendar=%s\n", calendarFor(r.Schedule))
|
|
// A fire missed because the machine was off happens when it comes back, rather than being
|
|
// skipped silently — which is the difference between a machine that was down and a schedule
|
|
// that quietly stopped.
|
|
b.WriteString("Persistent=true\n\n")
|
|
b.WriteString("[Install]\nWantedBy=timers.target\n")
|
|
return b.String()
|
|
}
|
|
|
|
// calendarFor turns five-field cron into what a systemd timer reads.
|
|
//
|
|
// minute hour day-of-month month day-of-week -> DayOfWeek Year-Month-Day Hour:Minute:Second
|
|
func calendarFor(cron string) string {
|
|
fields := strings.Fields(cron)
|
|
if len(fields) != 5 {
|
|
// Refused at validation, so this is unreachable — and returning something that would fire
|
|
// constantly is worse than returning something that never does.
|
|
return "*-*-* 00:00:00"
|
|
}
|
|
minute, hour, dom, month, dow := fields[0], fields[1], fields[2], fields[3], fields[4]
|
|
day := dow
|
|
if dow == "*" {
|
|
day = ""
|
|
} else {
|
|
day += " "
|
|
}
|
|
return fmt.Sprintf("%s*-%s-%s %s:%s:00", day, month, dom, hour, minute)
|
|
}
|
|
|
|
// unitValue renders one environment assignment so a unit file means what the declaration said.
|
|
//
|
|
// **Three things a unit file does to a value that nothing else does**, and a module's environment
|
|
// routinely contains all three — a generated password is arbitrary bytes.
|
|
//
|
|
// - `%` begins a specifier. `%H` is the hostname, `%i` the instance. A password containing one
|
|
// is silently replaced by something else, and the failure is an authentication error nobody
|
|
// can explain by looking at the declaration.
|
|
// - whitespace separates assignments. `Environment=K=a b` sets K to "a" and then tries to read
|
|
// "b" as another assignment.
|
|
// - a newline ends the line. What follows it is read as a unit DIRECTIVE, so a value carrying
|
|
// one could write ExecStart= and have the machine run something nobody declared.
|
|
//
|
|
// Quoted, with quotes and backslashes escaped and percent doubled. A container needs none of this
|
|
// because `--env` is passed through literally, which is why this had to be found here rather than
|
|
// noticed in both.
|
|
func unitValue(key, value string) string {
|
|
escaped := strings.NewReplacer(
|
|
`\`, `\\`,
|
|
`"`, `\"`,
|
|
"%", "%%",
|
|
).Replace(value)
|
|
return `"` + key + "=" + escaped + `"`
|
|
}
|