The last step of the first-node path, and the bundle now carries all of it: a container runtime, the store, a database per context, their schemas, the broker with a certificate it generated itself, and the control plane running. Then the machine enrols against the mesh on its own disk. It dials the broker over TLS, refuses anything but the pinned certificate, presents the one-time secret with a public key it generated, and is told the name the mesh has for it. Its specialness lasted two commands, which is what ADR 0004 asked for. The identity is saved only after the mesh says it knows this node. A node holding an identity the mesh never recorded would believe it had joined and be believed by nobody, which is worse than not joining because nothing looks wrong. An already-enrolled machine refuses a valid token rather than quietly acquiring a second identity, and a spent token is refused by the mesh. Both checked. Containers gained a network field. The control plane must reach the store and the broker on the machine it was raised on, before there is any mesh to arrange that; the alternative was publishing ports and guessing an address that works from inside a container, which fails in a worse way. The control plane talks to the broker over loopback in plaintext, deliberately. The TLS on 5671 exists so a node crossing a network can pin a certificate, not for a hop that never leaves the machine. Verified on a sealed lab machine: eleven resources applied from bare, the control plane consuming, a token issued from inside it, and the machine enrolled -- with the recorded public key matching what the host printed, the token marked spent, and the profile stored.
723 lines
24 KiB
Go
723 lines
24 KiB
Go
// Package apply makes a machine match a declaration.
|
|
//
|
|
// Three properties, each following a recorded decision, and each of them the difference
|
|
// between this and a script that writes files:
|
|
//
|
|
// - A failed step fails the apply (novox/hq ADR 0010). Not "logs and continues": a partial
|
|
// apply that reports success is the mesh's most expensive shape.
|
|
// - Every applier READS BACK. Setting a value is not evidence the value took.
|
|
// - What was applied is recorded after it works, never before (ADR 0018). A failed apply
|
|
// leaves the machine in whatever state it reached, and nothing must claim otherwise.
|
|
package apply
|
|
|
|
import (
|
|
"context"
|
|
"crypto/sha256"
|
|
"errors"
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"path/filepath"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/novox/mesh-host/internal/declaration"
|
|
"github.com/novox/mesh-host/internal/store"
|
|
"github.com/novox/mesh-host/internal/system"
|
|
)
|
|
|
|
// Runner executes a command. The real one is used everywhere outside unit tests; behaviour
|
|
// against a real system is tested alongside rather than mocked (novox/hq ADR 0017).
|
|
type Runner = system.Runner
|
|
|
|
// Outcome is what happened to one resource.
|
|
type Outcome struct {
|
|
ID string `json:"id"`
|
|
Type string `json:"type"`
|
|
Target string `json:"target"`
|
|
Action string `json:"action"` // created · updated · unchanged · removed
|
|
Detail string `json:"detail,omitempty"`
|
|
}
|
|
|
|
// Report is what an apply did, in the order it did it.
|
|
type Report struct {
|
|
Outcomes []Outcome `json:"outcomes"`
|
|
}
|
|
|
|
// Changed reports whether anything about the machine actually moved. An apply that changed
|
|
// nothing is the ordinary steady state, and saying so is not the same as saying it failed.
|
|
func (r Report) Changed() bool {
|
|
for _, o := range r.Outcomes {
|
|
if o.Action != "unchanged" {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// Error is a failure part-way through, carrying what had already been done.
|
|
//
|
|
// The outcomes matter as much as the message: the machine is in whatever state the apply
|
|
// reached, and the only honest thing to hand back is the list of what did happen.
|
|
type Error struct {
|
|
Resource string
|
|
Err error
|
|
Done Report
|
|
}
|
|
|
|
func (e *Error) Error() string {
|
|
return fmt.Sprintf("applying %q: %v\n\n%d resource(s) were applied before this and remain; "+
|
|
"the machine is in whatever state that left it.", e.Resource, e.Err, len(e.Done.Outcomes))
|
|
}
|
|
|
|
func (e *Error) Unwrap() error { return e.Err }
|
|
|
|
// Apply makes the machine match the declaration, and returns what it did.
|
|
//
|
|
// Removal happens FIRST, and the order is not arbitrary. A resource that leaves a declaration
|
|
// while another arrives at the same path is an ordinary rename: removing afterwards would
|
|
// delete the file that had just been written. Removing first risks losing the old state if the
|
|
// apply then fails — a recovery concern, where the other is a correctness one.
|
|
func Apply(
|
|
ctx context.Context,
|
|
sys system.System,
|
|
d *declaration.Declaration,
|
|
known store.State,
|
|
run Runner,
|
|
log func(string),
|
|
) (Report, store.State, error) {
|
|
if log == nil {
|
|
log = func(string) {}
|
|
}
|
|
report := Report{}
|
|
|
|
declared := map[string]bool{}
|
|
for _, r := range d.Resources {
|
|
declared[r.Identity()] = true
|
|
}
|
|
|
|
for _, orphan := range known.Orphans(declared) {
|
|
action, detail, err := remove(ctx, sys, orphan, run)
|
|
if err != nil {
|
|
return report, known, &Error{Resource: orphan.ID, Err: err, Done: report}
|
|
}
|
|
known.Forget(orphan.ID)
|
|
report.Outcomes = append(report.Outcomes, Outcome{
|
|
ID: orphan.ID, Type: orphan.Type, Target: orphan.Target,
|
|
Action: action, Detail: detail,
|
|
})
|
|
log(fmt.Sprintf(" %s %s (%s)", action, orphan.ID, orphan.Target))
|
|
}
|
|
|
|
for _, resource := range d.Resources {
|
|
outcome, err := applyOne(ctx, sys, resource, run)
|
|
if err != nil {
|
|
return report, known, &Error{Resource: resource.Identity(), Err: err, Done: report}
|
|
}
|
|
|
|
// Only now. The record follows the fact, never leads it.
|
|
known.Record(store.Applied{
|
|
ID: resource.Identity(), Type: string(resource.Kind()),
|
|
Target: outcome.Target, AppliedAt: time.Now().UTC(),
|
|
})
|
|
report.Outcomes = append(report.Outcomes, outcome)
|
|
if outcome.Action != "unchanged" {
|
|
log(fmt.Sprintf(" %s %s (%s)", outcome.Action, outcome.ID, outcome.Target))
|
|
}
|
|
}
|
|
return report, known, nil
|
|
}
|
|
|
|
func applyOne(ctx context.Context, sys system.System, r declaration.Resource, run Runner) (Outcome, error) {
|
|
switch res := r.(type) {
|
|
case *declaration.Directory:
|
|
return applyDirectory(res)
|
|
case *declaration.File:
|
|
return applyFile(res)
|
|
case *declaration.Service:
|
|
return applyService(ctx, sys, res, run)
|
|
case *declaration.Package:
|
|
return applyPackage(ctx, sys, res, run)
|
|
case *declaration.Container:
|
|
return applyContainer(ctx, res, run)
|
|
case *declaration.Action:
|
|
return applyAction(ctx, res, run)
|
|
default:
|
|
// Unreachable: the declaration refused this already. Present because "unreachable"
|
|
// stops being true the moment someone adds a kind and forgets this switch.
|
|
return Outcome{}, fmt.Errorf("no applier for type %q", r.Kind())
|
|
}
|
|
}
|
|
|
|
// begin starts an outcome from any resource, so the three facts a report needs are read from
|
|
// the resource itself rather than restated by each applier.
|
|
func begin(r declaration.Resource) Outcome {
|
|
return Outcome{ID: r.Identity(), Type: string(r.Kind()), Target: r.Target()}
|
|
}
|
|
|
|
func modeOf(spec string, fallback os.FileMode) (os.FileMode, error) {
|
|
if spec == "" {
|
|
return fallback, nil
|
|
}
|
|
parsed, err := strconv.ParseUint(spec, 8, 32)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("mode %q: %w", spec, err)
|
|
}
|
|
return os.FileMode(parsed), nil
|
|
}
|
|
|
|
func applyDirectory(r *declaration.Directory) (Outcome, error) {
|
|
out := begin(r)
|
|
mode, err := modeOf(r.Mode, 0o755)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
|
|
before, err := os.Stat(r.Path)
|
|
existed := err == nil
|
|
if err != nil && !errors.Is(err, os.ErrNotExist) {
|
|
return out, err
|
|
}
|
|
if existed && !before.IsDir() {
|
|
return out, fmt.Errorf("%s exists and is not a directory", r.Path)
|
|
}
|
|
|
|
if !existed {
|
|
if err := os.MkdirAll(r.Path, mode); err != nil {
|
|
return out, err
|
|
}
|
|
}
|
|
// Set explicitly even when it existed: MkdirAll applies the mode only on creation, and a
|
|
// permission set at creation is not a permission maintained — a lesson this repository
|
|
// already paid for once, with world-readable environment files.
|
|
if err := os.Chmod(r.Path, mode); err != nil {
|
|
return out, err
|
|
}
|
|
|
|
// Read back.
|
|
after, err := os.Stat(r.Path)
|
|
if err != nil {
|
|
return out, fmt.Errorf("made %s and cannot stat it: %w", r.Path, err)
|
|
}
|
|
if !after.IsDir() {
|
|
return out, fmt.Errorf("%s is not a directory after applying", r.Path)
|
|
}
|
|
if after.Mode().Perm() != mode.Perm() {
|
|
return out, fmt.Errorf("%s is mode %o after setting %o", r.Path, after.Mode().Perm(), mode.Perm())
|
|
}
|
|
|
|
out.Action = "unchanged"
|
|
if !existed {
|
|
out.Action = "created"
|
|
} else if before.Mode().Perm() != mode.Perm() {
|
|
out.Action = "updated"
|
|
out.Detail = fmt.Sprintf("mode %o to %o", before.Mode().Perm(), mode.Perm())
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
func applyFile(r *declaration.File) (Outcome, error) {
|
|
out := begin(r)
|
|
mode, err := modeOf(r.Mode, 0o644)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
|
|
existing, readErr := os.ReadFile(r.Path)
|
|
existed := readErr == nil
|
|
if readErr != nil && !errors.Is(readErr, os.ErrNotExist) {
|
|
return out, readErr
|
|
}
|
|
|
|
var beforeMode os.FileMode
|
|
if existed {
|
|
if info, err := os.Stat(r.Path); err == nil {
|
|
beforeMode = info.Mode().Perm()
|
|
}
|
|
}
|
|
|
|
contentSame := existed && string(existing) == r.Content
|
|
modeSame := existed && beforeMode == mode.Perm()
|
|
|
|
if !contentSame {
|
|
if err := os.MkdirAll(filepath.Dir(r.Path), 0o755); err != nil {
|
|
return out, err
|
|
}
|
|
if err := writeAtomically(r.Path, []byte(r.Content), mode); err != nil {
|
|
return out, err
|
|
}
|
|
} else if !modeSame {
|
|
if err := os.Chmod(r.Path, mode); err != nil {
|
|
return out, err
|
|
}
|
|
}
|
|
|
|
// Read back — the file, not the call that wrote it.
|
|
written, err := os.ReadFile(r.Path)
|
|
if err != nil {
|
|
return out, fmt.Errorf("wrote %s and cannot read it back: %w", r.Path, err)
|
|
}
|
|
if string(written) != r.Content {
|
|
return out, fmt.Errorf("%s does not contain what was declared after writing it", r.Path)
|
|
}
|
|
info, err := os.Stat(r.Path)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if info.Mode().Perm() != mode.Perm() {
|
|
return out, fmt.Errorf("%s is mode %o after setting %o", r.Path, info.Mode().Perm(), mode.Perm())
|
|
}
|
|
|
|
switch {
|
|
case !existed:
|
|
out.Action = "created"
|
|
case !contentSame && !modeSame:
|
|
out.Action = "updated"
|
|
out.Detail = "content and mode"
|
|
case !contentSame:
|
|
out.Action = "updated"
|
|
out.Detail = "content"
|
|
case !modeSame:
|
|
out.Action = "updated"
|
|
out.Detail = fmt.Sprintf("mode %o to %o", beforeMode, mode.Perm())
|
|
default:
|
|
out.Action = "unchanged"
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// writeAtomically writes through a temporary file in the same directory.
|
|
//
|
|
// A reader of a managed file must never see half of one. The mesh's own configuration is read
|
|
// by daemons that reload on change, so a torn write is a service reading a truncated config.
|
|
func writeAtomically(path string, content []byte, mode os.FileMode) error {
|
|
tmp, err := os.CreateTemp(filepath.Dir(path), ".mesh-host-*")
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer os.Remove(tmp.Name())
|
|
|
|
if _, err := tmp.Write(content); err != nil {
|
|
tmp.Close()
|
|
return err
|
|
}
|
|
if err := tmp.Sync(); err != nil {
|
|
tmp.Close()
|
|
return err
|
|
}
|
|
if err := tmp.Close(); err != nil {
|
|
return err
|
|
}
|
|
if err := os.Chmod(tmp.Name(), mode); err != nil {
|
|
return err
|
|
}
|
|
return os.Rename(tmp.Name(), path)
|
|
}
|
|
|
|
func applyService(ctx context.Context, sys system.System, r *declaration.Service, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
var changes []string
|
|
|
|
// Boot first. A unit asked to be running and enabled should survive this apply failing
|
|
// half way in the more useful direction: enabled-and-stopped comes back at the next boot,
|
|
// where running-and-disabled does not.
|
|
if r.Boot != "" {
|
|
bootBefore, err := sys.ServiceBoot(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if bootBefore != r.Boot {
|
|
if err := sys.SetServiceBoot(ctx, run, r.Unit, r.Boot); err != nil {
|
|
return out, fmt.Errorf("setting %s to %s at boot: %w", r.Unit, r.Boot, err)
|
|
}
|
|
bootAfter, err := sys.ServiceBoot(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if bootAfter != r.Boot {
|
|
return out, fmt.Errorf(
|
|
"%s was asked to be %s at boot and is %s", r.Unit, r.Boot, bootAfter)
|
|
}
|
|
changes = append(changes, "boot "+bootBefore+" to "+bootAfter)
|
|
}
|
|
}
|
|
|
|
before, err := sys.ServiceState(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if before != r.State {
|
|
if err := sys.SetServiceState(ctx, run, r.Unit, r.State); err != nil {
|
|
return out, fmt.Errorf("setting %s to %s: %w", r.Unit, r.State, err)
|
|
}
|
|
|
|
// Read back. A service manager accepting a command says the transaction was accepted,
|
|
// not that the unit is running — one that starts and immediately dies satisfies it.
|
|
after, err := sys.ServiceState(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if after != r.State {
|
|
return out, fmt.Errorf("%s was asked to be %s and is %s", r.Unit, r.State, after)
|
|
}
|
|
changes = append(changes, before+" to "+after)
|
|
}
|
|
|
|
if len(changes) == 0 {
|
|
out.Action = "unchanged"
|
|
out.Detail = before
|
|
return out, nil
|
|
}
|
|
out.Action = "updated"
|
|
out.Detail = strings.Join(changes, ", ")
|
|
return out, nil
|
|
}
|
|
|
|
// remove undoes one resource the host applied and the declaration no longer names, and reports
|
|
// what it actually did.
|
|
//
|
|
// Only ever called for something in the store, which is what bounds it: the host is
|
|
// authoritative over its own footprint and inert everywhere else (novox/hq ADR 0005).
|
|
//
|
|
// It returns the action rather than assuming "removed", because for half the vocabulary the
|
|
// honest word is "forgotten". A host that reported a package removed when it left the package
|
|
// installed would be describing an effect it declined to have.
|
|
func remove(ctx context.Context, sys system.System, a store.Applied, run Runner) (string, string, error) {
|
|
switch declaration.Type(a.Type) {
|
|
case declaration.TypeFile, declaration.TypeDirectory:
|
|
if err := os.RemoveAll(a.Target); err != nil {
|
|
return "", "", err
|
|
}
|
|
if _, err := os.Stat(a.Target); !errors.Is(err, os.ErrNotExist) {
|
|
return "", "", fmt.Errorf("%s is still there after removing it", a.Target)
|
|
}
|
|
return "removed", "no longer declared", nil
|
|
|
|
case declaration.TypeService:
|
|
// A unit that is no longer declared is stopped, not deleted. The host did not install
|
|
// it and does not own the unit file — only the state it put the unit into.
|
|
//
|
|
// A unit that no longer EXISTS is already in the state removal is trying to reach, and
|
|
// saying so matters: stopping it fails, and a failure here fails the whole apply. A
|
|
// host holding a record of an uninstalled unit would then be unable to apply anything,
|
|
// ever, with no way out but editing its state by hand. Removal is idempotent for the
|
|
// same reason `os.RemoveAll` is.
|
|
if _, err := sys.ServiceState(ctx, run, a.Target); err != nil {
|
|
if strings.Contains(err.Error(), "does not exist on this machine") {
|
|
return "forgotten", "the unit no longer exists", nil
|
|
}
|
|
return "", "", err
|
|
}
|
|
if err := sys.SetServiceState(ctx, run, a.Target, "stopped"); err != nil {
|
|
return "", "", fmt.Errorf("stopping %s: %w", a.Target, err)
|
|
}
|
|
return "removed", "stopped; the unit file is not the host's to delete", nil
|
|
|
|
case declaration.TypeContainer:
|
|
// The host CREATED this one, so the host removes it. That is the line: it removes what
|
|
// it made and leaves what it merely configured.
|
|
if _, err := run(ctx, "docker", "rm", "-f", a.Target); err != nil {
|
|
// Already gone is the state removal wants. Anything else is a real failure.
|
|
if _, alive := containerState(ctx, a.Target, run); alive == nil {
|
|
return "", "", fmt.Errorf("removing container %s: %w", a.Target, err)
|
|
}
|
|
}
|
|
if _, err := containerState(ctx, a.Target, run); err == nil {
|
|
return "", "", fmt.Errorf("container %s is still there after removing it", a.Target)
|
|
}
|
|
return "removed", "no longer declared", nil
|
|
|
|
case declaration.TypePackage:
|
|
// Deliberately not uninstalled, and this is a decision rather than an omission.
|
|
//
|
|
// The host cannot know what else on this machine needs the package. Uninstalling a
|
|
// container runtime because a declaration changed would stop every container on the
|
|
// node, and the machine may have had the package before the mesh ever saw it
|
|
// (novox/hq research 012: adopted, not installed). Undeclaring says "the mesh no
|
|
// longer requires this", which is not the same as "remove it".
|
|
return "forgotten", "left installed; the host does not uninstall what it cannot know is unused", nil
|
|
|
|
case declaration.TypeAction:
|
|
// An action has no footprint the host can undo — it ran, and whatever it did belongs
|
|
// to whatever it acted on.
|
|
return "forgotten", "an action leaves nothing the host owns", nil
|
|
|
|
default:
|
|
return "", "", fmt.Errorf("no way to remove a %q", a.Type)
|
|
}
|
|
}
|
|
|
|
// ExecRunner runs a real command, with stdin closed and output captured.
|
|
func ExecRunner(ctx context.Context, name string, args ...string) (string, error) {
|
|
cmd := exec.CommandContext(ctx, name, args...)
|
|
cmd.Stdin = nil
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
var exit *exec.ExitError
|
|
if errors.As(err, &exit) {
|
|
return string(out), fmt.Errorf("%s exited %d: %s",
|
|
name, exit.ExitCode(), strings.TrimSpace(string(exit.Stderr)))
|
|
}
|
|
return string(out), fmt.Errorf("%s: %w", name, err)
|
|
}
|
|
return string(out), nil
|
|
}
|
|
|
|
// applyPackage installs a package the machine does not have.
|
|
//
|
|
// It never upgrades and never removes. "Present" is the whole of what a package resource
|
|
// asserts, because version is the package manager's business and the mesh does not have a
|
|
// second opinion about it (novox/hq ADR 0005 — the host depends on nothing, and that includes
|
|
// not becoming a second package manager).
|
|
func applyPackage(ctx context.Context, sys system.System, r *declaration.Package, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
|
|
installed, err := sys.PackageInstalled(ctx, run, r.Package)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if installed {
|
|
out.Action = "unchanged"
|
|
out.Detail = "already installed"
|
|
return out, nil
|
|
}
|
|
|
|
if err := sys.InstallPackage(ctx, run, r.Package); err != nil {
|
|
return out, fmt.Errorf("installing %s: %w", r.Package, err)
|
|
}
|
|
|
|
// Read back. A package manager exiting zero says the transaction was accepted.
|
|
installed, err = sys.PackageInstalled(ctx, run, r.Package)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if !installed {
|
|
return out, fmt.Errorf(
|
|
"%s was installed without error and the package database does not have it", r.Package)
|
|
}
|
|
|
|
out.Action = "created"
|
|
return out, nil
|
|
}
|
|
|
|
// Labels the host puts on every container it creates.
|
|
//
|
|
// specLabel carries a digest of the declaration that made the container. It is what lets a
|
|
// reconcile answer "is this container the one the current declaration describes" without
|
|
// comparing every field the runtime reports — which cannot be done reliably, because a runtime
|
|
// normalises, defaults and reorders what it is given, and the differences that produces are
|
|
// indistinguishable from real drift.
|
|
const (
|
|
specLabel = "mesh-host.spec"
|
|
idLabel = "mesh-host.id"
|
|
)
|
|
|
|
// containerSpec is the identity of a declared container: everything that, if changed, means
|
|
// the running container is no longer what was asked for.
|
|
func containerSpec(r *declaration.Container) string {
|
|
keys := make([]string, 0, len(r.Env))
|
|
for k := range r.Env {
|
|
keys = append(keys, k)
|
|
}
|
|
sort.Strings(keys)
|
|
|
|
var b strings.Builder
|
|
b.WriteString(r.Image + "\n" + r.Name + "\n")
|
|
for _, k := range keys {
|
|
b.WriteString("env " + k + "=" + r.Env[k] + "\n")
|
|
}
|
|
for _, p := range r.Ports {
|
|
b.WriteString("port " + p + "\n")
|
|
}
|
|
for _, v := range r.Volumes {
|
|
b.WriteString("volume " + v + "\n")
|
|
}
|
|
for _, a := range r.Args {
|
|
b.WriteString("arg " + a + "\n")
|
|
}
|
|
return fmt.Sprintf("%x", sha256.Sum256([]byte(b.String())))
|
|
}
|
|
|
|
// containerState reports whether a container is running and which spec made it.
|
|
// The error means the container does not exist.
|
|
func containerState(ctx context.Context, name string, run Runner) (state struct {
|
|
Running bool
|
|
Spec string
|
|
}, err error) {
|
|
out, err := run(ctx, "docker", "inspect", "--format",
|
|
"{{.State.Running}}\t{{index .Config.Labels \""+specLabel+"\"}}", name)
|
|
if err != nil {
|
|
return state, fmt.Errorf("no container named %s", name)
|
|
}
|
|
running, spec, _ := strings.Cut(strings.TrimSpace(out), "\t")
|
|
state.Running = running == "true"
|
|
state.Spec = strings.TrimSpace(spec)
|
|
return state, nil
|
|
}
|
|
|
|
// applyContainer makes the declared container the one that is running.
|
|
//
|
|
// There is no "update" for a container: a container's configuration is fixed when it is
|
|
// created, so any change is a replacement. Saying that plainly is better than a partial
|
|
// in-place update that leaves the running thing half-declared.
|
|
func applyContainer(ctx context.Context, r *declaration.Container, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
want := containerSpec(r)
|
|
|
|
cri, err := containerRuntime(ctx, run)
|
|
if err != nil {
|
|
return out, fmt.Errorf("%w, so nothing can be said about %q", err, r.Name)
|
|
}
|
|
|
|
before, err := containerState(ctx, r.Name, run)
|
|
existed := err == nil
|
|
|
|
switch {
|
|
case existed && before.Spec == want && before.Running:
|
|
out.Action = "unchanged"
|
|
return out, nil
|
|
case existed:
|
|
if _, err := run(ctx, cri, "rm", "-f", r.Name); err != nil {
|
|
return out, fmt.Errorf("replacing container %s: %w", r.Name, err)
|
|
}
|
|
}
|
|
|
|
args := []string{"run", "--detach", "--name", r.Name, "--restart", "unless-stopped"}
|
|
if r.Network != "" {
|
|
args = append(args, "--network", r.Network)
|
|
}
|
|
args = append(args,
|
|
"--label", specLabel+"="+want, "--label", idLabel+"="+r.ID)
|
|
for _, k := range sortedKeys(r.Env) {
|
|
args = append(args, "--env", k+"="+r.Env[k])
|
|
}
|
|
for _, p := range r.Ports {
|
|
args = append(args, "--publish", p)
|
|
}
|
|
for _, v := range r.Volumes {
|
|
args = append(args, "--volume", v)
|
|
}
|
|
args = append(args, r.Image)
|
|
args = append(args, r.Args...)
|
|
|
|
if _, err := run(ctx, cri, args...); err != nil {
|
|
return out, fmt.Errorf("starting container %s: %w", r.Name, err)
|
|
}
|
|
|
|
// Read back. `docker run --detach` returning an id says the container was created, not
|
|
// that it is still running — a container whose entrypoint exits immediately satisfies the
|
|
// command exactly as one that came up does.
|
|
after, err := containerState(ctx, r.Name, run)
|
|
if err != nil {
|
|
return out, fmt.Errorf("started container %s and it is not there: %w", r.Name, err)
|
|
}
|
|
if !after.Running {
|
|
return out, fmt.Errorf(
|
|
"container %s was started and is not running. It exited; ask the runtime for its "+
|
|
"logs", r.Name)
|
|
}
|
|
if after.Spec != want {
|
|
return out, fmt.Errorf("container %s is not the one that was declared after creating it", r.Name)
|
|
}
|
|
|
|
out.Action = "created"
|
|
if existed {
|
|
out.Action = "updated"
|
|
out.Detail = "replaced; a container's configuration is fixed when it is created"
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
func sortedKeys(m map[string]string) []string {
|
|
keys := make([]string, 0, len(m))
|
|
for k := range m {
|
|
keys = append(keys, k)
|
|
}
|
|
sort.Strings(keys)
|
|
return keys
|
|
}
|
|
|
|
// applyAction runs something the bundle declared, and never learns what it means.
|
|
//
|
|
// Verify does double duty, and that is the design rather than a convenience: it is both the
|
|
// idempotency check and the read-back. Running it first is how the host knows whether there is
|
|
// anything to do — it does not know what a database is, so "is the database there" is a
|
|
// question only the declaration can ask. Running it again afterwards is how the host knows the
|
|
// command had the effect it claimed (novox/hq ADR 0005).
|
|
func applyAction(ctx context.Context, r *declaration.Action, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
|
|
if _, err := runAction(ctx, r, r.Verify, run); err == nil {
|
|
out.Action = "unchanged"
|
|
out.Detail = "already true"
|
|
return out, nil
|
|
}
|
|
|
|
if _, err := runAction(ctx, r, r.Command, run); err != nil {
|
|
return out, fmt.Errorf("running the action: %w", err)
|
|
}
|
|
|
|
if _, err := runAction(ctx, r, r.Verify, run); err != nil {
|
|
return out, fmt.Errorf(
|
|
"the action ran without error and its own verify still fails: %w\n\n"+
|
|
"The command reported success and the thing it was for did not happen, which "+
|
|
"is exactly what verify exists to catch", err)
|
|
}
|
|
|
|
out.Action = "created"
|
|
out.Detail = "verify was false and is now true"
|
|
return out, nil
|
|
}
|
|
|
|
// runAction runs one of an action's command lines, on the machine or inside a container.
|
|
func runAction(ctx context.Context, r *declaration.Action, argv []string, run Runner) (string, error) {
|
|
if len(argv) == 0 {
|
|
return "", errors.New("no command")
|
|
}
|
|
if r.In != "" {
|
|
return run(ctx, "docker", append([]string{"exec", r.In}, argv...)...)
|
|
}
|
|
return run(ctx, argv[0], argv[1:]...)
|
|
}
|
|
|
|
// Container runtimes the host knows how to ask.
|
|
//
|
|
// Two, because two exist on machines the mesh runs on. The list is short on purpose: each entry
|
|
// is a claim that its probe and its CLI have been checked, not that a binary of that name might
|
|
// work (novox/hq ADR 0005).
|
|
//
|
|
// The probe differs and the rest does not, which is what makes this a lookup rather than an
|
|
// interface. `docker info --format {{.ServerVersion}}` fails on podman — the field does not
|
|
// exist in its report — while `run`, `inspect --format` and `rm -f` are identical, including
|
|
// docker's own Go template syntax for reading state and labels.
|
|
var containerRuntimes = []struct {
|
|
command string
|
|
// probe asks the runtime for its version in the form THAT runtime understands. It must
|
|
// prove the runtime is FUNCTIONING, never that a binary is on disk
|
|
// (novox/hq 04-ISSUES/007).
|
|
probe []string
|
|
}{
|
|
{command: "docker", probe: []string{"info", "--format", "{{.ServerVersion}}"}},
|
|
{command: "podman", probe: []string{"info", "--format", "{{.Version.Version}}"}},
|
|
}
|
|
|
|
// containerRuntime returns the runtime this machine actually has, or says there is none.
|
|
//
|
|
// Detected rather than declared, because a machine already carrying one keeps it: adoption
|
|
// takes over what is there rather than replacing it (novox/hq research 012). Which runtime a
|
|
// machine has is reported upward in the profile; what to install on a machine with none is the
|
|
// control plane's decision, not this one's.
|
|
func containerRuntime(ctx context.Context, run Runner) (string, error) {
|
|
var tried []string
|
|
for _, rt := range containerRuntimes {
|
|
if _, err := run(ctx, rt.command, rt.probe...); err == nil {
|
|
return rt.command, nil
|
|
}
|
|
tried = append(tried, rt.command)
|
|
}
|
|
return "", fmt.Errorf(
|
|
"no container runtime answers on this machine (tried %s)", strings.Join(tried, ", "))
|
|
}
|