Files
mesh-host/internal/apply/container.go
T
jschoubben c80f671203 apply: container and network resources, over the container runtime
Extends the applier past the filesystem to the two types the workloads
need: a container and the private network it joins. The workloads are
the bulk of what a cutover re-declares (research 009), so this is what
makes a workload manifest actually appliable.

- container: run/reconcile/remove over the runtime. Up to date means a
  container that is ours (a spec-hash label matches this exact
  declaration) AND running; anything else — a changed spec, a stopped
  container, or a foreign one the old control plane left by that name —
  is recreated into ours. Safe because a container carries no state:
  its data is in bind-mounted directories declared separately, and
  recreating it never touches them. Read-back asks the runtime whether
  it is actually running on the declared spec, because 'started' only
  means the runtime returned.
- network: create if absent, adopt if present, remove only what it
  created.
- The runtime is driven through a Runner, faked in unit tests and
  exercised for real in a smoke test that stands a container up, proves
  idempotency, and tears it down — skipped, never failed, where the
  runtime is absent.

The store's per-resource reference generalises from a path to a ref:
a path for files and directories, a name for containers and networks.

Verified end to end through the binary: a container on a bind mount,
then dropped from the declaration — the container is removed and the
data directory survives, which is the migration property itself.

Still deferred: sealed secrets, and package/service/archive/user/action
— refused whole until built, never half-applied.

Claude-Session: https://claude.ai/code/session_01LrgweAeERJYBg88c5cKDzF
2026-09-02 22:48:27 +02:00

220 lines
7.7 KiB
Go

package apply
import (
"crypto/sha256"
"encoding/hex"
"encoding/json"
"fmt"
"os/exec"
"strings"
)
// containerRuntime is the runtime the mesh uses. Named rather than assumed, so the one place that
// would change for podman is here — and profile already detects that a container-runtime is
// present before any container resource is placed on a node.
const containerRuntime = "docker"
// specLabel carries a hash of the desired container spec. It is what lets a re-apply tell an
// up-to-date container from one that must be recreated, without parsing the runtime's own view of
// env, ports and mounts and hoping the comparison matches field for field. The host wrote the
// hash; the host trusts the hash.
const specLabel = "mesh-host.spec"
// Runner executes a container-runtime command. Replaceable in tests, and exercised against the
// real runtime as well (novox/hq ADR 0034): a test that only fakes the runtime asserts the fake
// behaves as expected, so the real path is smoke-tested too.
type Runner func(name string, args ...string) (string, error)
func execRunner(name string, args ...string) (string, error) {
out, err := exec.Command(name, args...).CombinedOutput()
if err != nil {
return string(out), fmt.Errorf("%s %s: %w: %s",
name, strings.Join(args, " "), err, strings.TrimSpace(string(out)))
}
return string(out), nil
}
// --- network ---
type networkApplier struct{ run Runner }
func (networkApplier) Type() string { return "network" }
func (n networkApplier) Apply(r Resource) (bool, error) {
name := r.stringField("name")
if _, err := n.run(containerRuntime, "network", "inspect", name); err == nil {
return false, nil // already present — adopted, not created
}
if _, err := n.run(containerRuntime, "network", "create", name); err != nil {
return false, fmt.Errorf("creating network %s: %w", name, err)
}
// Read back: the network must now be inspectable, or the create did not take.
if _, err := n.run(containerRuntime, "network", "inspect", name); err != nil {
return true, fmt.Errorf("network %s did not take: %w", name, err)
}
return true, nil
}
func (n networkApplier) Remove(rec Record) error {
if _, err := n.run(containerRuntime, "network", "rm", rec.Ref); err != nil {
if alreadyGone(err) {
return nil
}
return fmt.Errorf("removing network %s: %w", rec.Ref, err)
}
return nil
}
// --- container ---
type containerApplier struct{ run Runner }
func (containerApplier) Type() string { return "container" }
func (c containerApplier) Apply(r Resource) (bool, error) {
name := r.stringField("name")
image := r.stringField("image")
if image == "" {
return false, fmt.Errorf("container %q names no image", name)
}
hash := specHash(r)
exists, running, curHash := c.inspect(name)
// A container is up to date only when it is ours (its spec label matches this exact spec) AND
// it is running. A foreign container by the same name — one the old control plane started, with
// no label — does not match, and is recreated into ours. That is safe: a container carries no
// state, its data lives in bind-mounted directories declared separately, and recreating it does
// not touch them.
upToDate := exists && running && curHash == hash
if upToDate {
return false, nil // present and correct; this apply created nothing
}
if exists {
if _, err := c.run(containerRuntime, "rm", "-f", name); err != nil {
return false, fmt.Errorf("replacing container %s: %w", name, err)
}
}
if _, err := c.run(containerRuntime, runArgs(r, hash)...); err != nil {
return true, fmt.Errorf("starting container %s: %w", name, err)
}
// Read back: the container must now be running, on this exact spec. "Service started" only
// means the runtime returned — the host asks whether it is actually up (novox/hq
// troubleshooting/service-started-is-not-ready).
exists, running, curHash = c.inspect(name)
if !exists || !running {
return true, fmt.Errorf("container %s did not come up", name)
}
if curHash != hash {
return true, fmt.Errorf("container %s came up on a spec that is not the one declared", name)
}
return true, nil
}
func (c containerApplier) Remove(rec Record) error {
if _, err := c.run(containerRuntime, "rm", "-f", rec.Ref); err != nil {
if alreadyGone(err) {
return nil
}
return fmt.Errorf("removing container %s: %w", rec.Ref, err)
}
return nil
}
// alreadyGone reports whether a removal failed only because the thing was not there — which is
// success, not failure. The runtime phrases it variously ("No such container", "no such object",
// "not found") across versions, so the match is lenient and case-insensitive.
func alreadyGone(err error) bool {
msg := strings.ToLower(err.Error())
return strings.Contains(msg, "no such") || strings.Contains(msg, "not found")
}
// inspect reports whether a container by this name exists, whether it is running, and the spec
// hash it was labelled with (empty for a container the host did not label).
func (c containerApplier) inspect(name string) (exists, running bool, hash string) {
out, err := c.run(containerRuntime, "inspect", "-f",
"{{.State.Running}}|{{index .Config.Labels \""+specLabel+"\"}}", name)
if err != nil {
return false, false, ""
}
parts := strings.SplitN(strings.TrimSpace(out), "|", 2)
running = parts[0] == "true"
if len(parts) == 2 && parts[1] != "<no value>" {
hash = parts[1]
}
return true, running, hash
}
// runArgs builds the `docker run` invocation for r, labelled with its spec hash.
func runArgs(r Resource, hash string) []string {
args := []string{"run", "-d", "--name", r.stringField("name"), "--label", specLabel + "=" + hash}
if net := r.stringField("network"); net != "" {
args = append(args, "--network", net)
}
for _, ef := range r.stringSlice("env-file") {
args = append(args, "--env-file", ef)
}
// Env is emitted in the JSON-sorted order specHash also uses, so the invocation is stable.
env := r.stringMap("env")
for _, k := range sortedKeys(env) {
args = append(args, "-e", k+"="+env[k])
}
for _, p := range r.stringSlice("ports") {
args = append(args, "-p", p)
}
for _, v := range r.stringSlice("volumes") {
args = append(args, "-v", v)
}
for _, h := range r.stringSlice("hosts") {
args = append(args, "--add-host", h)
}
args = append(args, r.stringField("image"))
args = append(args, r.stringSlice("args")...)
return args
}
// specHash is a stable fingerprint of everything that decides whether a running container matches
// what is declared. Marshalled through a struct so the field set is explicit, and json.Marshal
// sorts map keys, so the same declaration always hashes the same.
func specHash(r Resource) string {
type spec struct {
Name string `json:"name"`
Image string `json:"image"`
Network string `json:"network"`
Env map[string]string `json:"env"`
EnvFile []string `json:"env_file"`
Ports []string `json:"ports"`
Volumes []string `json:"volumes"`
Hosts []string `json:"hosts"`
Args []string `json:"args"`
}
b, _ := json.Marshal(spec{
Name: r.stringField("name"),
Image: r.stringField("image"),
Network: r.stringField("network"),
Env: r.stringMap("env"),
EnvFile: r.stringSlice("env-file"),
Ports: r.stringSlice("ports"),
Volumes: r.stringSlice("volumes"),
Hosts: r.stringSlice("hosts"),
Args: r.stringSlice("args"),
})
sum := sha256.Sum256(b)
return hex.EncodeToString(sum[:])[:16]
}
func sortedKeys(m map[string]string) []string {
out := make([]string, 0, len(m))
for k := range m {
out = append(out, k)
}
// small n; insertion order does not matter, only that it is stable and sorted
for i := 1; i < len(out); i++ {
for j := i; j > 0 && out[j-1] > out[j]; j-- {
out[j-1], out[j] = out[j], out[j-1]
}
}
return out
}