96 comments across the two repos named records that no longer exist. Each now points at the consolidated record that holds its reasoning -- ADR 0034 (a test defends a decision) is 0017, the eight host records are 0005, the four lab records are 0016. Worth noting for next time: these are references from outside HQ, so renumbering there is not free. It cost 38 files here.
719 lines
24 KiB
Go
719 lines
24 KiB
Go
// Package apply makes a machine match a declaration.
|
|
//
|
|
// Three properties, each following a recorded decision, and each of them the difference
|
|
// between this and a script that writes files:
|
|
//
|
|
// - A failed step fails the apply (novox/hq ADR 0010). Not "logs and continues": a partial
|
|
// apply that reports success is the mesh's most expensive shape.
|
|
// - Every applier READS BACK. Setting a value is not evidence the value took.
|
|
// - What was applied is recorded after it works, never before (ADR 0018). A failed apply
|
|
// leaves the machine in whatever state it reached, and nothing must claim otherwise.
|
|
package apply
|
|
|
|
import (
|
|
"context"
|
|
"crypto/sha256"
|
|
"errors"
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"path/filepath"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/novox/mesh-host/internal/declaration"
|
|
"github.com/novox/mesh-host/internal/store"
|
|
"github.com/novox/mesh-host/internal/system"
|
|
)
|
|
|
|
// Runner executes a command. The real one is used everywhere outside unit tests; behaviour
|
|
// against a real system is tested alongside rather than mocked (novox/hq ADR 0017).
|
|
type Runner = system.Runner
|
|
|
|
// Outcome is what happened to one resource.
|
|
type Outcome struct {
|
|
ID string `json:"id"`
|
|
Type string `json:"type"`
|
|
Target string `json:"target"`
|
|
Action string `json:"action"` // created · updated · unchanged · removed
|
|
Detail string `json:"detail,omitempty"`
|
|
}
|
|
|
|
// Report is what an apply did, in the order it did it.
|
|
type Report struct {
|
|
Outcomes []Outcome `json:"outcomes"`
|
|
}
|
|
|
|
// Changed reports whether anything about the machine actually moved. An apply that changed
|
|
// nothing is the ordinary steady state, and saying so is not the same as saying it failed.
|
|
func (r Report) Changed() bool {
|
|
for _, o := range r.Outcomes {
|
|
if o.Action != "unchanged" {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// Error is a failure part-way through, carrying what had already been done.
|
|
//
|
|
// The outcomes matter as much as the message: the machine is in whatever state the apply
|
|
// reached, and the only honest thing to hand back is the list of what did happen.
|
|
type Error struct {
|
|
Resource string
|
|
Err error
|
|
Done Report
|
|
}
|
|
|
|
func (e *Error) Error() string {
|
|
return fmt.Sprintf("applying %q: %v\n\n%d resource(s) were applied before this and remain; "+
|
|
"the machine is in whatever state that left it.", e.Resource, e.Err, len(e.Done.Outcomes))
|
|
}
|
|
|
|
func (e *Error) Unwrap() error { return e.Err }
|
|
|
|
// Apply makes the machine match the declaration, and returns what it did.
|
|
//
|
|
// Removal happens FIRST, and the order is not arbitrary. A resource that leaves a declaration
|
|
// while another arrives at the same path is an ordinary rename: removing afterwards would
|
|
// delete the file that had just been written. Removing first risks losing the old state if the
|
|
// apply then fails — a recovery concern, where the other is a correctness one.
|
|
func Apply(
|
|
ctx context.Context,
|
|
sys system.System,
|
|
d *declaration.Declaration,
|
|
known store.State,
|
|
run Runner,
|
|
log func(string),
|
|
) (Report, store.State, error) {
|
|
if log == nil {
|
|
log = func(string) {}
|
|
}
|
|
report := Report{}
|
|
|
|
declared := map[string]bool{}
|
|
for _, r := range d.Resources {
|
|
declared[r.Identity()] = true
|
|
}
|
|
|
|
for _, orphan := range known.Orphans(declared) {
|
|
action, detail, err := remove(ctx, sys, orphan, run)
|
|
if err != nil {
|
|
return report, known, &Error{Resource: orphan.ID, Err: err, Done: report}
|
|
}
|
|
known.Forget(orphan.ID)
|
|
report.Outcomes = append(report.Outcomes, Outcome{
|
|
ID: orphan.ID, Type: orphan.Type, Target: orphan.Target,
|
|
Action: action, Detail: detail,
|
|
})
|
|
log(fmt.Sprintf(" %s %s (%s)", action, orphan.ID, orphan.Target))
|
|
}
|
|
|
|
for _, resource := range d.Resources {
|
|
outcome, err := applyOne(ctx, sys, resource, run)
|
|
if err != nil {
|
|
return report, known, &Error{Resource: resource.Identity(), Err: err, Done: report}
|
|
}
|
|
|
|
// Only now. The record follows the fact, never leads it.
|
|
known.Record(store.Applied{
|
|
ID: resource.Identity(), Type: string(resource.Kind()),
|
|
Target: outcome.Target, AppliedAt: time.Now().UTC(),
|
|
})
|
|
report.Outcomes = append(report.Outcomes, outcome)
|
|
if outcome.Action != "unchanged" {
|
|
log(fmt.Sprintf(" %s %s (%s)", outcome.Action, outcome.ID, outcome.Target))
|
|
}
|
|
}
|
|
return report, known, nil
|
|
}
|
|
|
|
func applyOne(ctx context.Context, sys system.System, r declaration.Resource, run Runner) (Outcome, error) {
|
|
switch res := r.(type) {
|
|
case *declaration.Directory:
|
|
return applyDirectory(res)
|
|
case *declaration.File:
|
|
return applyFile(res)
|
|
case *declaration.Service:
|
|
return applyService(ctx, sys, res, run)
|
|
case *declaration.Package:
|
|
return applyPackage(ctx, sys, res, run)
|
|
case *declaration.Container:
|
|
return applyContainer(ctx, res, run)
|
|
case *declaration.Action:
|
|
return applyAction(ctx, res, run)
|
|
default:
|
|
// Unreachable: the declaration refused this already. Present because "unreachable"
|
|
// stops being true the moment someone adds a kind and forgets this switch.
|
|
return Outcome{}, fmt.Errorf("no applier for type %q", r.Kind())
|
|
}
|
|
}
|
|
|
|
// begin starts an outcome from any resource, so the three facts a report needs are read from
|
|
// the resource itself rather than restated by each applier.
|
|
func begin(r declaration.Resource) Outcome {
|
|
return Outcome{ID: r.Identity(), Type: string(r.Kind()), Target: r.Target()}
|
|
}
|
|
|
|
func modeOf(spec string, fallback os.FileMode) (os.FileMode, error) {
|
|
if spec == "" {
|
|
return fallback, nil
|
|
}
|
|
parsed, err := strconv.ParseUint(spec, 8, 32)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("mode %q: %w", spec, err)
|
|
}
|
|
return os.FileMode(parsed), nil
|
|
}
|
|
|
|
func applyDirectory(r *declaration.Directory) (Outcome, error) {
|
|
out := begin(r)
|
|
mode, err := modeOf(r.Mode, 0o755)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
|
|
before, err := os.Stat(r.Path)
|
|
existed := err == nil
|
|
if err != nil && !errors.Is(err, os.ErrNotExist) {
|
|
return out, err
|
|
}
|
|
if existed && !before.IsDir() {
|
|
return out, fmt.Errorf("%s exists and is not a directory", r.Path)
|
|
}
|
|
|
|
if !existed {
|
|
if err := os.MkdirAll(r.Path, mode); err != nil {
|
|
return out, err
|
|
}
|
|
}
|
|
// Set explicitly even when it existed: MkdirAll applies the mode only on creation, and a
|
|
// permission set at creation is not a permission maintained — a lesson this repository
|
|
// already paid for once, with world-readable environment files.
|
|
if err := os.Chmod(r.Path, mode); err != nil {
|
|
return out, err
|
|
}
|
|
|
|
// Read back.
|
|
after, err := os.Stat(r.Path)
|
|
if err != nil {
|
|
return out, fmt.Errorf("made %s and cannot stat it: %w", r.Path, err)
|
|
}
|
|
if !after.IsDir() {
|
|
return out, fmt.Errorf("%s is not a directory after applying", r.Path)
|
|
}
|
|
if after.Mode().Perm() != mode.Perm() {
|
|
return out, fmt.Errorf("%s is mode %o after setting %o", r.Path, after.Mode().Perm(), mode.Perm())
|
|
}
|
|
|
|
out.Action = "unchanged"
|
|
if !existed {
|
|
out.Action = "created"
|
|
} else if before.Mode().Perm() != mode.Perm() {
|
|
out.Action = "updated"
|
|
out.Detail = fmt.Sprintf("mode %o to %o", before.Mode().Perm(), mode.Perm())
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
func applyFile(r *declaration.File) (Outcome, error) {
|
|
out := begin(r)
|
|
mode, err := modeOf(r.Mode, 0o644)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
|
|
existing, readErr := os.ReadFile(r.Path)
|
|
existed := readErr == nil
|
|
if readErr != nil && !errors.Is(readErr, os.ErrNotExist) {
|
|
return out, readErr
|
|
}
|
|
|
|
var beforeMode os.FileMode
|
|
if existed {
|
|
if info, err := os.Stat(r.Path); err == nil {
|
|
beforeMode = info.Mode().Perm()
|
|
}
|
|
}
|
|
|
|
contentSame := existed && string(existing) == r.Content
|
|
modeSame := existed && beforeMode == mode.Perm()
|
|
|
|
if !contentSame {
|
|
if err := os.MkdirAll(filepath.Dir(r.Path), 0o755); err != nil {
|
|
return out, err
|
|
}
|
|
if err := writeAtomically(r.Path, []byte(r.Content), mode); err != nil {
|
|
return out, err
|
|
}
|
|
} else if !modeSame {
|
|
if err := os.Chmod(r.Path, mode); err != nil {
|
|
return out, err
|
|
}
|
|
}
|
|
|
|
// Read back — the file, not the call that wrote it.
|
|
written, err := os.ReadFile(r.Path)
|
|
if err != nil {
|
|
return out, fmt.Errorf("wrote %s and cannot read it back: %w", r.Path, err)
|
|
}
|
|
if string(written) != r.Content {
|
|
return out, fmt.Errorf("%s does not contain what was declared after writing it", r.Path)
|
|
}
|
|
info, err := os.Stat(r.Path)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if info.Mode().Perm() != mode.Perm() {
|
|
return out, fmt.Errorf("%s is mode %o after setting %o", r.Path, info.Mode().Perm(), mode.Perm())
|
|
}
|
|
|
|
switch {
|
|
case !existed:
|
|
out.Action = "created"
|
|
case !contentSame && !modeSame:
|
|
out.Action = "updated"
|
|
out.Detail = "content and mode"
|
|
case !contentSame:
|
|
out.Action = "updated"
|
|
out.Detail = "content"
|
|
case !modeSame:
|
|
out.Action = "updated"
|
|
out.Detail = fmt.Sprintf("mode %o to %o", beforeMode, mode.Perm())
|
|
default:
|
|
out.Action = "unchanged"
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// writeAtomically writes through a temporary file in the same directory.
|
|
//
|
|
// A reader of a managed file must never see half of one. The mesh's own configuration is read
|
|
// by daemons that reload on change, so a torn write is a service reading a truncated config.
|
|
func writeAtomically(path string, content []byte, mode os.FileMode) error {
|
|
tmp, err := os.CreateTemp(filepath.Dir(path), ".mesh-host-*")
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer os.Remove(tmp.Name())
|
|
|
|
if _, err := tmp.Write(content); err != nil {
|
|
tmp.Close()
|
|
return err
|
|
}
|
|
if err := tmp.Sync(); err != nil {
|
|
tmp.Close()
|
|
return err
|
|
}
|
|
if err := tmp.Close(); err != nil {
|
|
return err
|
|
}
|
|
if err := os.Chmod(tmp.Name(), mode); err != nil {
|
|
return err
|
|
}
|
|
return os.Rename(tmp.Name(), path)
|
|
}
|
|
|
|
func applyService(ctx context.Context, sys system.System, r *declaration.Service, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
var changes []string
|
|
|
|
// Boot first. A unit asked to be running and enabled should survive this apply failing
|
|
// half way in the more useful direction: enabled-and-stopped comes back at the next boot,
|
|
// where running-and-disabled does not.
|
|
if r.Boot != "" {
|
|
bootBefore, err := sys.ServiceBoot(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if bootBefore != r.Boot {
|
|
if err := sys.SetServiceBoot(ctx, run, r.Unit, r.Boot); err != nil {
|
|
return out, fmt.Errorf("setting %s to %s at boot: %w", r.Unit, r.Boot, err)
|
|
}
|
|
bootAfter, err := sys.ServiceBoot(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if bootAfter != r.Boot {
|
|
return out, fmt.Errorf(
|
|
"%s was asked to be %s at boot and is %s", r.Unit, r.Boot, bootAfter)
|
|
}
|
|
changes = append(changes, "boot "+bootBefore+" to "+bootAfter)
|
|
}
|
|
}
|
|
|
|
before, err := sys.ServiceState(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if before != r.State {
|
|
if err := sys.SetServiceState(ctx, run, r.Unit, r.State); err != nil {
|
|
return out, fmt.Errorf("setting %s to %s: %w", r.Unit, r.State, err)
|
|
}
|
|
|
|
// Read back. A service manager accepting a command says the transaction was accepted,
|
|
// not that the unit is running — one that starts and immediately dies satisfies it.
|
|
after, err := sys.ServiceState(ctx, run, r.Unit)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if after != r.State {
|
|
return out, fmt.Errorf("%s was asked to be %s and is %s", r.Unit, r.State, after)
|
|
}
|
|
changes = append(changes, before+" to "+after)
|
|
}
|
|
|
|
if len(changes) == 0 {
|
|
out.Action = "unchanged"
|
|
out.Detail = before
|
|
return out, nil
|
|
}
|
|
out.Action = "updated"
|
|
out.Detail = strings.Join(changes, ", ")
|
|
return out, nil
|
|
}
|
|
|
|
// remove undoes one resource the host applied and the declaration no longer names, and reports
|
|
// what it actually did.
|
|
//
|
|
// Only ever called for something in the store, which is what bounds it: the host is
|
|
// authoritative over its own footprint and inert everywhere else (novox/hq ADR 0005).
|
|
//
|
|
// It returns the action rather than assuming "removed", because for half the vocabulary the
|
|
// honest word is "forgotten". A host that reported a package removed when it left the package
|
|
// installed would be describing an effect it declined to have.
|
|
func remove(ctx context.Context, sys system.System, a store.Applied, run Runner) (string, string, error) {
|
|
switch declaration.Type(a.Type) {
|
|
case declaration.TypeFile, declaration.TypeDirectory:
|
|
if err := os.RemoveAll(a.Target); err != nil {
|
|
return "", "", err
|
|
}
|
|
if _, err := os.Stat(a.Target); !errors.Is(err, os.ErrNotExist) {
|
|
return "", "", fmt.Errorf("%s is still there after removing it", a.Target)
|
|
}
|
|
return "removed", "no longer declared", nil
|
|
|
|
case declaration.TypeService:
|
|
// A unit that is no longer declared is stopped, not deleted. The host did not install
|
|
// it and does not own the unit file — only the state it put the unit into.
|
|
//
|
|
// A unit that no longer EXISTS is already in the state removal is trying to reach, and
|
|
// saying so matters: stopping it fails, and a failure here fails the whole apply. A
|
|
// host holding a record of an uninstalled unit would then be unable to apply anything,
|
|
// ever, with no way out but editing its state by hand. Removal is idempotent for the
|
|
// same reason `os.RemoveAll` is.
|
|
if _, err := sys.ServiceState(ctx, run, a.Target); err != nil {
|
|
if strings.Contains(err.Error(), "does not exist on this machine") {
|
|
return "forgotten", "the unit no longer exists", nil
|
|
}
|
|
return "", "", err
|
|
}
|
|
if err := sys.SetServiceState(ctx, run, a.Target, "stopped"); err != nil {
|
|
return "", "", fmt.Errorf("stopping %s: %w", a.Target, err)
|
|
}
|
|
return "removed", "stopped; the unit file is not the host's to delete", nil
|
|
|
|
case declaration.TypeContainer:
|
|
// The host CREATED this one, so the host removes it. That is the line: it removes what
|
|
// it made and leaves what it merely configured.
|
|
if _, err := run(ctx, "docker", "rm", "-f", a.Target); err != nil {
|
|
// Already gone is the state removal wants. Anything else is a real failure.
|
|
if _, alive := containerState(ctx, a.Target, run); alive == nil {
|
|
return "", "", fmt.Errorf("removing container %s: %w", a.Target, err)
|
|
}
|
|
}
|
|
if _, err := containerState(ctx, a.Target, run); err == nil {
|
|
return "", "", fmt.Errorf("container %s is still there after removing it", a.Target)
|
|
}
|
|
return "removed", "no longer declared", nil
|
|
|
|
case declaration.TypePackage:
|
|
// Deliberately not uninstalled, and this is a decision rather than an omission.
|
|
//
|
|
// The host cannot know what else on this machine needs the package. Uninstalling a
|
|
// container runtime because a declaration changed would stop every container on the
|
|
// node, and the machine may have had the package before the mesh ever saw it
|
|
// (novox/hq research 012: adopted, not installed). Undeclaring says "the mesh no
|
|
// longer requires this", which is not the same as "remove it".
|
|
return "forgotten", "left installed; the host does not uninstall what it cannot know is unused", nil
|
|
|
|
case declaration.TypeAction:
|
|
// An action has no footprint the host can undo — it ran, and whatever it did belongs
|
|
// to whatever it acted on.
|
|
return "forgotten", "an action leaves nothing the host owns", nil
|
|
|
|
default:
|
|
return "", "", fmt.Errorf("no way to remove a %q", a.Type)
|
|
}
|
|
}
|
|
|
|
// ExecRunner runs a real command, with stdin closed and output captured.
|
|
func ExecRunner(ctx context.Context, name string, args ...string) (string, error) {
|
|
cmd := exec.CommandContext(ctx, name, args...)
|
|
cmd.Stdin = nil
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
var exit *exec.ExitError
|
|
if errors.As(err, &exit) {
|
|
return string(out), fmt.Errorf("%s exited %d: %s",
|
|
name, exit.ExitCode(), strings.TrimSpace(string(exit.Stderr)))
|
|
}
|
|
return string(out), fmt.Errorf("%s: %w", name, err)
|
|
}
|
|
return string(out), nil
|
|
}
|
|
|
|
// applyPackage installs a package the machine does not have.
|
|
//
|
|
// It never upgrades and never removes. "Present" is the whole of what a package resource
|
|
// asserts, because version is the package manager's business and the mesh does not have a
|
|
// second opinion about it (novox/hq ADR 0005 — the host depends on nothing, and that includes
|
|
// not becoming a second package manager).
|
|
func applyPackage(ctx context.Context, sys system.System, r *declaration.Package, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
|
|
installed, err := sys.PackageInstalled(ctx, run, r.Package)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if installed {
|
|
out.Action = "unchanged"
|
|
out.Detail = "already installed"
|
|
return out, nil
|
|
}
|
|
|
|
if err := sys.InstallPackage(ctx, run, r.Package); err != nil {
|
|
return out, fmt.Errorf("installing %s: %w", r.Package, err)
|
|
}
|
|
|
|
// Read back. A package manager exiting zero says the transaction was accepted.
|
|
installed, err = sys.PackageInstalled(ctx, run, r.Package)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
if !installed {
|
|
return out, fmt.Errorf(
|
|
"%s was installed without error and the package database does not have it", r.Package)
|
|
}
|
|
|
|
out.Action = "created"
|
|
return out, nil
|
|
}
|
|
|
|
// Labels the host puts on every container it creates.
|
|
//
|
|
// specLabel carries a digest of the declaration that made the container. It is what lets a
|
|
// reconcile answer "is this container the one the current declaration describes" without
|
|
// comparing every field the runtime reports — which cannot be done reliably, because a runtime
|
|
// normalises, defaults and reorders what it is given, and the differences that produces are
|
|
// indistinguishable from real drift.
|
|
const (
|
|
specLabel = "mesh-host.spec"
|
|
idLabel = "mesh-host.id"
|
|
)
|
|
|
|
// containerSpec is the identity of a declared container: everything that, if changed, means
|
|
// the running container is no longer what was asked for.
|
|
func containerSpec(r *declaration.Container) string {
|
|
keys := make([]string, 0, len(r.Env))
|
|
for k := range r.Env {
|
|
keys = append(keys, k)
|
|
}
|
|
sort.Strings(keys)
|
|
|
|
var b strings.Builder
|
|
b.WriteString(r.Image + "\n" + r.Name + "\n")
|
|
for _, k := range keys {
|
|
b.WriteString("env " + k + "=" + r.Env[k] + "\n")
|
|
}
|
|
for _, p := range r.Ports {
|
|
b.WriteString("port " + p + "\n")
|
|
}
|
|
for _, v := range r.Volumes {
|
|
b.WriteString("volume " + v + "\n")
|
|
}
|
|
for _, a := range r.Args {
|
|
b.WriteString("arg " + a + "\n")
|
|
}
|
|
return fmt.Sprintf("%x", sha256.Sum256([]byte(b.String())))
|
|
}
|
|
|
|
// containerState reports whether a container is running and which spec made it.
|
|
// The error means the container does not exist.
|
|
func containerState(ctx context.Context, name string, run Runner) (state struct {
|
|
Running bool
|
|
Spec string
|
|
}, err error) {
|
|
out, err := run(ctx, "docker", "inspect", "--format",
|
|
"{{.State.Running}}\t{{index .Config.Labels \""+specLabel+"\"}}", name)
|
|
if err != nil {
|
|
return state, fmt.Errorf("no container named %s", name)
|
|
}
|
|
running, spec, _ := strings.Cut(strings.TrimSpace(out), "\t")
|
|
state.Running = running == "true"
|
|
state.Spec = strings.TrimSpace(spec)
|
|
return state, nil
|
|
}
|
|
|
|
// applyContainer makes the declared container the one that is running.
|
|
//
|
|
// There is no "update" for a container: a container's configuration is fixed when it is
|
|
// created, so any change is a replacement. Saying that plainly is better than a partial
|
|
// in-place update that leaves the running thing half-declared.
|
|
func applyContainer(ctx context.Context, r *declaration.Container, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
want := containerSpec(r)
|
|
|
|
cri, err := containerRuntime(ctx, run)
|
|
if err != nil {
|
|
return out, fmt.Errorf("%w, so nothing can be said about %q", err, r.Name)
|
|
}
|
|
|
|
before, err := containerState(ctx, r.Name, run)
|
|
existed := err == nil
|
|
|
|
switch {
|
|
case existed && before.Spec == want && before.Running:
|
|
out.Action = "unchanged"
|
|
return out, nil
|
|
case existed:
|
|
if _, err := run(ctx, cri, "rm", "-f", r.Name); err != nil {
|
|
return out, fmt.Errorf("replacing container %s: %w", r.Name, err)
|
|
}
|
|
}
|
|
|
|
args := []string{"run", "--detach", "--name", r.Name, "--restart", "unless-stopped",
|
|
"--label", specLabel + "=" + want, "--label", idLabel + "=" + r.ID}
|
|
for _, k := range sortedKeys(r.Env) {
|
|
args = append(args, "--env", k+"="+r.Env[k])
|
|
}
|
|
for _, p := range r.Ports {
|
|
args = append(args, "--publish", p)
|
|
}
|
|
for _, v := range r.Volumes {
|
|
args = append(args, "--volume", v)
|
|
}
|
|
args = append(args, r.Image)
|
|
args = append(args, r.Args...)
|
|
|
|
if _, err := run(ctx, cri, args...); err != nil {
|
|
return out, fmt.Errorf("starting container %s: %w", r.Name, err)
|
|
}
|
|
|
|
// Read back. `docker run --detach` returning an id says the container was created, not
|
|
// that it is still running — a container whose entrypoint exits immediately satisfies the
|
|
// command exactly as one that came up does.
|
|
after, err := containerState(ctx, r.Name, run)
|
|
if err != nil {
|
|
return out, fmt.Errorf("started container %s and it is not there: %w", r.Name, err)
|
|
}
|
|
if !after.Running {
|
|
return out, fmt.Errorf(
|
|
"container %s was started and is not running. It exited; ask the runtime for its "+
|
|
"logs", r.Name)
|
|
}
|
|
if after.Spec != want {
|
|
return out, fmt.Errorf("container %s is not the one that was declared after creating it", r.Name)
|
|
}
|
|
|
|
out.Action = "created"
|
|
if existed {
|
|
out.Action = "updated"
|
|
out.Detail = "replaced; a container's configuration is fixed when it is created"
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
func sortedKeys(m map[string]string) []string {
|
|
keys := make([]string, 0, len(m))
|
|
for k := range m {
|
|
keys = append(keys, k)
|
|
}
|
|
sort.Strings(keys)
|
|
return keys
|
|
}
|
|
|
|
// applyAction runs something the bundle declared, and never learns what it means.
|
|
//
|
|
// Verify does double duty, and that is the design rather than a convenience: it is both the
|
|
// idempotency check and the read-back. Running it first is how the host knows whether there is
|
|
// anything to do — it does not know what a database is, so "is the database there" is a
|
|
// question only the declaration can ask. Running it again afterwards is how the host knows the
|
|
// command had the effect it claimed (novox/hq ADR 0005).
|
|
func applyAction(ctx context.Context, r *declaration.Action, run Runner) (Outcome, error) {
|
|
out := begin(r)
|
|
|
|
if _, err := runAction(ctx, r, r.Verify, run); err == nil {
|
|
out.Action = "unchanged"
|
|
out.Detail = "already true"
|
|
return out, nil
|
|
}
|
|
|
|
if _, err := runAction(ctx, r, r.Command, run); err != nil {
|
|
return out, fmt.Errorf("running the action: %w", err)
|
|
}
|
|
|
|
if _, err := runAction(ctx, r, r.Verify, run); err != nil {
|
|
return out, fmt.Errorf(
|
|
"the action ran without error and its own verify still fails: %w\n\n"+
|
|
"The command reported success and the thing it was for did not happen, which "+
|
|
"is exactly what verify exists to catch", err)
|
|
}
|
|
|
|
out.Action = "created"
|
|
out.Detail = "verify was false and is now true"
|
|
return out, nil
|
|
}
|
|
|
|
// runAction runs one of an action's command lines, on the machine or inside a container.
|
|
func runAction(ctx context.Context, r *declaration.Action, argv []string, run Runner) (string, error) {
|
|
if len(argv) == 0 {
|
|
return "", errors.New("no command")
|
|
}
|
|
if r.In != "" {
|
|
return run(ctx, "docker", append([]string{"exec", r.In}, argv...)...)
|
|
}
|
|
return run(ctx, argv[0], argv[1:]...)
|
|
}
|
|
|
|
// Container runtimes the host knows how to ask.
|
|
//
|
|
// Two, because two exist on machines the mesh runs on. The list is short on purpose: each entry
|
|
// is a claim that its probe and its CLI have been checked, not that a binary of that name might
|
|
// work (novox/hq ADR 0005).
|
|
//
|
|
// The probe differs and the rest does not, which is what makes this a lookup rather than an
|
|
// interface. `docker info --format {{.ServerVersion}}` fails on podman — the field does not
|
|
// exist in its report — while `run`, `inspect --format` and `rm -f` are identical, including
|
|
// docker's own Go template syntax for reading state and labels.
|
|
var containerRuntimes = []struct {
|
|
command string
|
|
// probe asks the runtime for its version in the form THAT runtime understands. It must
|
|
// prove the runtime is FUNCTIONING, never that a binary is on disk
|
|
// (novox/hq 04-ISSUES/007).
|
|
probe []string
|
|
}{
|
|
{command: "docker", probe: []string{"info", "--format", "{{.ServerVersion}}"}},
|
|
{command: "podman", probe: []string{"info", "--format", "{{.Version.Version}}"}},
|
|
}
|
|
|
|
// containerRuntime returns the runtime this machine actually has, or says there is none.
|
|
//
|
|
// Detected rather than declared, because a machine already carrying one keeps it: adoption
|
|
// takes over what is there rather than replacing it (novox/hq research 012). Which runtime a
|
|
// machine has is reported upward in the profile; what to install on a machine with none is the
|
|
// control plane's decision, not this one's.
|
|
func containerRuntime(ctx context.Context, run Runner) (string, error) {
|
|
var tried []string
|
|
for _, rt := range containerRuntimes {
|
|
if _, err := run(ctx, rt.command, rt.probe...); err == nil {
|
|
return rt.command, nil
|
|
}
|
|
tried = append(tried, rt.command)
|
|
}
|
|
return "", fmt.Errorf(
|
|
"no container runtime answers on this machine (tried %s)", strings.Join(tried, ", "))
|
|
}
|