Claims node-container-runtime (ADR 0207). Owns the packages, the socket and a weekly prune of dangling images and unused build cache. Serves 18 tools over every container, marking the mesh's. daemon.json, docker.service and the docker group are left to a proposed change: dnsmasq and zsh declare them today, and the controller refuses a second declaration (README).
1065 lines
35 KiB
Go
1065 lines
35 KiB
Go
package main
|
|
|
|
// The container runtime's own command line, asked by the operator account the node's tool runtime
|
|
// runs as (novox/hq ADR 0175 §4). That account is in the docker group on every machine, but a
|
|
// process keeps the groups it started with: a runtime started before the account joined the group
|
|
// is refused by the daemon's socket. So a refused socket is asked again through `sudo -n`, as the
|
|
// service manager's and the packet filter's acts are, and a refusal there is named by how it failed.
|
|
//
|
|
// A container the mesh holds carries the host's label (MeshLabel); its value names the assignment
|
|
// (`<module>.<resource>`). Every answer says which containers are the mesh's, because the host
|
|
// restores a held container to what its declaration says at its next apply (novox/hq ADR 0166).
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"os"
|
|
"regexp"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// MeshLabel is the label the host puts on every container it creates (mesh-host internal/apply).
|
|
const MeshLabel = "mesh-host.id"
|
|
|
|
// DaemonFile is the runtime's configuration file.
|
|
const DaemonFile = "/etc/docker/daemon.json"
|
|
|
|
// Client asks the runtime.
|
|
type Client struct {
|
|
Run Runner
|
|
UID int
|
|
ReadFile func(string) ([]byte, error)
|
|
Now func() time.Time
|
|
}
|
|
|
|
// NewClient is the client the bundle serves with.
|
|
func NewClient() *Client {
|
|
return &Client{Run: ExecRunner, UID: os.Getuid(), ReadFile: os.ReadFile, Now: time.Now}
|
|
}
|
|
|
|
var socketRefused = regexp.MustCompile(`(?i)permission denied.*docker.*sock|docker\.sock.*permission denied`)
|
|
|
|
// docker runs one docker command and answers its stdout, or an error naming what went wrong.
|
|
func (c *Client) docker(ctx context.Context, args ...string) (string, error) {
|
|
r := c.Run(ctx, "docker", args...)
|
|
program := "docker"
|
|
if r.Status != 0 && r.Err == "" && c.UID != 0 && socketRefused.MatchString(r.Stderr) {
|
|
program = "sudo"
|
|
r = c.Run(ctx, "sudo", append([]string{"-n", "docker"}, args...)...)
|
|
}
|
|
if r.Status == 0 && r.Err == "" {
|
|
return r.Stdout, nil
|
|
}
|
|
return "", failure(args, program, r)
|
|
}
|
|
|
|
func failure(args []string, program string, r Ran) error {
|
|
verb := "docker"
|
|
if len(args) > 0 {
|
|
verb += " " + args[0]
|
|
}
|
|
said := strings.TrimSpace(r.Stderr + "\n" + r.Stdout)
|
|
switch {
|
|
case r.Err == "ENOENT" && program == "sudo":
|
|
return fmt.Errorf("the runtime's socket refused this account, and sudo is not installed here to escalate with")
|
|
case r.Err == "ENOENT":
|
|
return fmt.Errorf("docker is not installed on this machine")
|
|
case r.Err != "":
|
|
return fmt.Errorf("%s did not answer: %s (the daemon may still be doing it)", verb, r.Err)
|
|
case program == "sudo" && regexp.MustCompile(`(?m)^sudo:`).MatchString(said):
|
|
return fmt.Errorf("the runtime's socket refused this account (not in the docker group, or not since the tool runtime started) and it may not escalate without a prompt: %s", firstLine(said))
|
|
case strings.Contains(said, "Cannot connect to the Docker daemon"):
|
|
return fmt.Errorf("the docker daemon is not answering on this machine: %s", firstLine(said))
|
|
case strings.Contains(said, "No such container") || strings.Contains(said, "No such object"):
|
|
return fmt.Errorf("%s", firstLine(said))
|
|
}
|
|
if l := firstLine(said); l != "" {
|
|
return fmt.Errorf("%s failed (%d): %s", verb, r.Status, l)
|
|
}
|
|
return fmt.Errorf("%s failed with status %d", verb, r.Status)
|
|
}
|
|
|
|
var refPattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9_.:/@+-]*$`)
|
|
|
|
// Ref is a container's, image's, network's or volume's name or id as an argument: never something
|
|
// docker would read as an option, which under sudo would be root's option.
|
|
func Ref(s string) (string, error) {
|
|
s = strings.TrimSpace(s)
|
|
if !refPattern.MatchString(s) || len(s) > 256 {
|
|
return "", fmt.Errorf("%q is not a name or id docker knows things by", s)
|
|
}
|
|
return s, nil
|
|
}
|
|
|
|
// lines splits output into its non-empty lines.
|
|
func lines(s string) []string {
|
|
var out []string
|
|
for _, l := range strings.Split(s, "\n") {
|
|
if l = strings.TrimSpace(l); l != "" {
|
|
out = append(out, l)
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// jsonLines decodes one JSON object per line, as `--format '{{json .}}'` prints them.
|
|
func jsonLines[T any](s string) ([]T, error) {
|
|
out := []T{}
|
|
for _, l := range lines(s) {
|
|
var v T
|
|
if err := json.Unmarshal([]byte(l), &v); err != nil {
|
|
return nil, fmt.Errorf("docker answered a line that is not JSON: %.120s", l)
|
|
}
|
|
out = append(out, v)
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// Bytes reads docker's human sizes ("55.63GB", "33.2MiB", "0B"): decimal units as docker's own
|
|
// HumanSize prints them, binary ones where it says so. -1 when it is not a size.
|
|
func Bytes(s string) int64 {
|
|
s = strings.TrimSpace(s)
|
|
if i := strings.Index(s, " ("); i > 0 {
|
|
s = s[:i]
|
|
}
|
|
m := regexp.MustCompile(`^([0-9.]+)\s*([A-Za-z]*)$`).FindStringSubmatch(s)
|
|
if m == nil {
|
|
return -1
|
|
}
|
|
n, err := strconv.ParseFloat(m[1], 64)
|
|
if err != nil {
|
|
return -1
|
|
}
|
|
units := map[string]float64{"": 1, "B": 1, "kB": 1e3, "KB": 1e3, "MB": 1e6, "GB": 1e9, "TB": 1e12, "PB": 1e15,
|
|
"KiB": 1 << 10, "MiB": 1 << 20, "GiB": 1 << 30, "TiB": 1 << 40}
|
|
u, ok := units[m[2]]
|
|
if !ok {
|
|
return -1
|
|
}
|
|
return int64(n * u)
|
|
}
|
|
|
|
// ---- containers -------------------------------------------------------------------------------
|
|
|
|
type inspected struct {
|
|
ID string `json:"Id"`
|
|
Name string
|
|
Created string
|
|
Image string
|
|
Config struct {
|
|
Image string
|
|
Labels map[string]string
|
|
}
|
|
State struct {
|
|
Status string
|
|
Running bool
|
|
Restarting bool
|
|
OOMKilled bool
|
|
ExitCode int
|
|
StartedAt string
|
|
FinishedAt string
|
|
Health *struct{ Status string }
|
|
}
|
|
RestartCount int
|
|
HostConfig struct {
|
|
RestartPolicy struct{ Name string }
|
|
NetworkMode string
|
|
Privileged bool
|
|
}
|
|
NetworkSettings struct {
|
|
Ports map[string][]struct {
|
|
HostIP string `json:"HostIp"`
|
|
HostPort string
|
|
}
|
|
}
|
|
Mounts []struct {
|
|
Type string
|
|
Name string
|
|
Source string
|
|
Destination string
|
|
RW bool
|
|
}
|
|
}
|
|
|
|
// Mount is one thing a container mounts.
|
|
type Mount struct {
|
|
Type string `json:"type"`
|
|
Name string `json:"name,omitempty"`
|
|
Source string `json:"source,omitempty"`
|
|
Destination string `json:"destination"`
|
|
ReadOnly bool `json:"read_only,omitempty"`
|
|
}
|
|
|
|
// Container is one container as every tool here answers it.
|
|
type Container struct {
|
|
ID string `json:"id"`
|
|
Name string `json:"name"`
|
|
Image string `json:"image"`
|
|
ImageID string `json:"image_id"`
|
|
State string `json:"state"`
|
|
Health string `json:"health,omitempty"`
|
|
ExitCode int `json:"exit_code"`
|
|
OOMKilled bool `json:"oom_killed,omitempty"`
|
|
Created string `json:"created"`
|
|
StartedAt string `json:"started_at,omitempty"`
|
|
FinishedAt string `json:"finished_at,omitempty"`
|
|
Restarts int `json:"restarts"`
|
|
RestartPolicy string `json:"restart_policy,omitempty"`
|
|
Network string `json:"network_mode,omitempty"`
|
|
Privileged bool `json:"privileged,omitempty"`
|
|
MeshHeld bool `json:"mesh_held"`
|
|
HeldBy string `json:"held_by,omitempty"`
|
|
Module string `json:"module,omitempty"`
|
|
Compose string `json:"compose_project,omitempty"`
|
|
ComposeDir string `json:"compose_dir,omitempty"`
|
|
Ports []string `json:"ports"`
|
|
Mounts []Mount `json:"mounts"`
|
|
}
|
|
|
|
func summary(i inspected) Container {
|
|
c := Container{
|
|
ID: shortID(i.ID), Name: strings.TrimPrefix(i.Name, "/"), Image: i.Config.Image, ImageID: i.Image,
|
|
State: i.State.Status, ExitCode: i.State.ExitCode, OOMKilled: i.State.OOMKilled, Created: i.Created,
|
|
Restarts: i.RestartCount, RestartPolicy: i.HostConfig.RestartPolicy.Name, Network: i.HostConfig.NetworkMode,
|
|
Privileged: i.HostConfig.Privileged, Ports: []string{}, Mounts: []Mount{},
|
|
}
|
|
if !zeroTime(i.State.StartedAt) {
|
|
c.StartedAt = i.State.StartedAt
|
|
}
|
|
if !zeroTime(i.State.FinishedAt) && !i.State.Running {
|
|
c.FinishedAt = i.State.FinishedAt
|
|
}
|
|
if i.State.Health != nil {
|
|
c.Health = i.State.Health.Status
|
|
}
|
|
if v, ok := i.Config.Labels[MeshLabel]; ok {
|
|
c.MeshHeld, c.HeldBy = true, v
|
|
c.Module, _, _ = strings.Cut(v, ".")
|
|
}
|
|
c.Compose = i.Config.Labels["com.docker.compose.project"]
|
|
c.ComposeDir = i.Config.Labels["com.docker.compose.project.working_dir"]
|
|
for port, binds := range i.NetworkSettings.Ports {
|
|
for _, b := range binds {
|
|
c.Ports = append(c.Ports, fmt.Sprintf("%s:%s->%s", b.HostIP, b.HostPort, port))
|
|
}
|
|
}
|
|
sort.Strings(c.Ports)
|
|
for _, m := range i.Mounts {
|
|
mm := Mount{Type: m.Type, Destination: m.Destination, ReadOnly: !m.RW}
|
|
if m.Type == "volume" {
|
|
mm.Name = m.Name
|
|
} else {
|
|
mm.Source = m.Source
|
|
}
|
|
c.Mounts = append(c.Mounts, mm)
|
|
}
|
|
return c
|
|
}
|
|
|
|
func zeroTime(s string) bool { return s == "" || strings.HasPrefix(s, "0001-01-01") }
|
|
|
|
func shortID(id string) string {
|
|
id = strings.TrimPrefix(id, "sha256:")
|
|
if len(id) > 12 {
|
|
return id[:12]
|
|
}
|
|
return id
|
|
}
|
|
|
|
// inspectAll is every container on the machine, the mesh's and every other.
|
|
func (c *Client) inspectAll(ctx context.Context) ([]inspected, error) {
|
|
ids, err := c.docker(ctx, "ps", "--all", "--quiet", "--no-trunc")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
all := lines(ids)
|
|
if len(all) == 0 {
|
|
return []inspected{}, nil
|
|
}
|
|
out, err := c.docker(ctx, append([]string{"container", "inspect"}, all...)...)
|
|
if err != nil {
|
|
// A container removed between the two calls fails the whole inspect; ask once more.
|
|
if strings.Contains(err.Error(), "No such") {
|
|
return c.inspectAll(ctx)
|
|
}
|
|
return nil, err
|
|
}
|
|
var got []inspected
|
|
if err := json.Unmarshal([]byte(out), &got); err != nil {
|
|
return nil, fmt.Errorf("docker inspect answered something that is not JSON: %v", err)
|
|
}
|
|
return got, nil
|
|
}
|
|
|
|
// Containers is every container, narrowed as asked.
|
|
func (c *Client) Containers(ctx context.Context, held, state, match string) ([]Container, error) {
|
|
all, err := c.inspectAll(ctx)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := []Container{}
|
|
for _, i := range all {
|
|
s := summary(i)
|
|
switch held {
|
|
case "", "all":
|
|
case "mesh":
|
|
if !s.MeshHeld {
|
|
continue
|
|
}
|
|
case "other":
|
|
if s.MeshHeld {
|
|
continue
|
|
}
|
|
default:
|
|
return nil, fmt.Errorf("held %q: \"all\", \"mesh\" or \"other\"", held)
|
|
}
|
|
if state != "" && s.State != state {
|
|
continue
|
|
}
|
|
if match != "" && !strings.Contains(s.Name, match) && !strings.Contains(s.Image, match) {
|
|
continue
|
|
}
|
|
out = append(out, s)
|
|
}
|
|
sort.Slice(out, func(a, b int) bool { return out[a].Name < out[b].Name })
|
|
return out, nil
|
|
}
|
|
|
|
// Inspect is one container whole, as docker inspects it, with every environment variable's value
|
|
// left out (names kept): a container's environment is where its passwords are.
|
|
func (c *Client) Inspect(ctx context.Context, ref string) (map[string]any, error) {
|
|
ref, err := Ref(ref)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out, err := c.docker(ctx, "container", "inspect", ref)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var got []map[string]any
|
|
if err := json.Unmarshal([]byte(out), &got); err != nil || len(got) != 1 {
|
|
return nil, fmt.Errorf("docker inspect answered something that is not one container")
|
|
}
|
|
obj := got[0]
|
|
if cfg, ok := obj["Config"].(map[string]any); ok {
|
|
if env, ok := cfg["Env"].([]any); ok {
|
|
names := []string{}
|
|
for _, e := range env {
|
|
name, _, _ := strings.Cut(fmt.Sprint(e), "=")
|
|
names = append(names, name)
|
|
}
|
|
cfg["Env"] = names
|
|
cfg["EnvValues"] = "left out: a container's environment holds its secrets"
|
|
}
|
|
if labels, ok := cfg["Labels"].(map[string]any); ok {
|
|
if v, ok := labels[MeshLabel]; ok {
|
|
obj["mesh_held"], obj["held_by"] = true, v
|
|
} else {
|
|
obj["mesh_held"] = false
|
|
}
|
|
}
|
|
}
|
|
return obj, nil
|
|
}
|
|
|
|
// Logs is the last lines a container wrote, both streams in the order they were written.
|
|
func (c *Client) Logs(ctx context.Context, ref string, tail int, since string) (map[string]any, error) {
|
|
ref, err := Ref(ref)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
args := []string{"logs", "--timestamps", "--tail", strconv.Itoa(tail)}
|
|
if since != "" {
|
|
if !regexp.MustCompile(`^[0-9]+[smhd]?$|^\d{4}-\d{2}-\d{2}`).MatchString(since) {
|
|
return nil, fmt.Errorf("since %q: a duration such as 30m or 2h, or a time such as 2026-10-04T10:00:00", since)
|
|
}
|
|
args = append(args, "--since", since)
|
|
}
|
|
args = append(args, ref)
|
|
r := c.Run(ctx, "docker", args...)
|
|
program := "docker"
|
|
if r.Status != 0 && r.Err == "" && c.UID != 0 && socketRefused.MatchString(r.Stderr) {
|
|
program = "sudo"
|
|
r = c.Run(ctx, "sudo", append([]string{"-n", "docker"}, args...)...)
|
|
}
|
|
if r.Status != 0 || r.Err != "" {
|
|
return nil, failure(args, program, r)
|
|
}
|
|
// Both streams carry the container's lines, each led by its timestamp, so they merge in order.
|
|
all := append(lines(r.Stdout), lines(r.Stderr)...)
|
|
sort.SliceStable(all, func(a, b int) bool { return all[a] < all[b] })
|
|
if len(all) > tail {
|
|
all = all[len(all)-tail:]
|
|
}
|
|
const most = 4096
|
|
for i, l := range all {
|
|
if len(l) > most {
|
|
all[i] = l[:most] + "…"
|
|
}
|
|
}
|
|
return map[string]any{"container": ref, "lines": all, "count": len(all)}, nil
|
|
}
|
|
|
|
// Stat is one container's use of the machine now.
|
|
type Stat struct {
|
|
Name string `json:"name"`
|
|
ID string `json:"id"`
|
|
CPU string `json:"cpu"`
|
|
Memory string `json:"memory"`
|
|
MemPerc string `json:"memory_percent"`
|
|
MemBytes int64 `json:"memory_bytes"`
|
|
NetIO string `json:"net_io"`
|
|
BlockIO string `json:"block_io"`
|
|
PIDs string `json:"pids"`
|
|
}
|
|
|
|
// Stats is a snapshot of every running container's use, or one's; the heaviest first.
|
|
func (c *Client) Stats(ctx context.Context, ref string) ([]Stat, error) {
|
|
args := []string{"stats", "--no-stream", "--format", "{{json .}}"}
|
|
if ref != "" {
|
|
r, err := Ref(ref)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
args = append(args, r)
|
|
}
|
|
out, err := c.docker(ctx, args...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
raw, err := jsonLines[map[string]string](out)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
stats := []Stat{}
|
|
for _, s := range raw {
|
|
used, _, _ := strings.Cut(s["MemUsage"], " / ")
|
|
stats = append(stats, Stat{Name: s["Name"], ID: s["ID"], CPU: s["CPUPerc"], Memory: s["MemUsage"], MemPerc: s["MemPerc"],
|
|
MemBytes: Bytes(used), NetIO: s["NetIO"], BlockIO: s["BlockIO"], PIDs: s["PIDs"]})
|
|
}
|
|
sort.Slice(stats, func(a, b int) bool { return stats[a].MemBytes > stats[b].MemBytes })
|
|
return stats, nil
|
|
}
|
|
|
|
// Act starts, stops or restarts one container and answers its state after. A container the mesh
|
|
// holds is acted on too — the host restores what its declaration says at its next apply, and the
|
|
// answer says so (novox/hq ADR 0166).
|
|
func (c *Client) Act(ctx context.Context, verb, ref string) (map[string]any, error) {
|
|
ref, err := Ref(ref)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
args := []string{verb}
|
|
switch verb {
|
|
case "start":
|
|
case "stop", "restart":
|
|
// Ten seconds to stop before it is killed: inside the call's own bound.
|
|
args = append(args, "--time", "10")
|
|
default:
|
|
return nil, fmt.Errorf("%q is not start, stop or restart", verb)
|
|
}
|
|
if _, err := c.docker(ctx, append(args, ref)...); err != nil {
|
|
return nil, err
|
|
}
|
|
out, err := c.docker(ctx, "container", "inspect", ref)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var got []inspected
|
|
if err := json.Unmarshal([]byte(out), &got); err != nil || len(got) != 1 {
|
|
return nil, fmt.Errorf("docker inspect answered something that is not one container")
|
|
}
|
|
s := summary(got[0])
|
|
answer := map[string]any{"container": s.Name, "verb": verb, "ok": true, "state": s.State, "mesh_held": s.MeshHeld}
|
|
if s.MeshHeld {
|
|
answer["held_by"] = s.HeldBy
|
|
answer["note"] = fmt.Sprintf("the mesh holds this container (%s): the host restores what its declaration says at its next apply", s.HeldBy)
|
|
}
|
|
return answer, nil
|
|
}
|
|
|
|
// Top is the processes running in one container.
|
|
func (c *Client) Top(ctx context.Context, ref string) (map[string]any, error) {
|
|
ref, err := Ref(ref)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out, err := c.docker(ctx, "top", ref, "-eo", "pid,user,etime,pcpu,rss,args")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
rows := lines(out)
|
|
procs := []map[string]string{}
|
|
for _, l := range rows[min(1, len(rows)):] {
|
|
f := strings.Fields(l)
|
|
if len(f) < 6 {
|
|
continue
|
|
}
|
|
procs = append(procs, map[string]string{"pid": f[0], "user": f[1], "elapsed": f[2], "cpu": f[3], "rss_kb": f[4], "command": strings.Join(f[5:], " ")})
|
|
}
|
|
return map[string]any{"container": ref, "processes": procs}, nil
|
|
}
|
|
|
|
// Problems is every container that is not well: unhealthy, restarting, dead, killed for memory,
|
|
// or exited with a failure.
|
|
func (c *Client) Problems(ctx context.Context) ([]map[string]any, error) {
|
|
all, err := c.Containers(ctx, "", "", "")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := []map[string]any{}
|
|
for _, s := range all {
|
|
var why []string
|
|
if s.Health == "unhealthy" {
|
|
why = append(why, "unhealthy")
|
|
}
|
|
if s.State == "restarting" {
|
|
why = append(why, "restarting")
|
|
}
|
|
if s.State == "dead" {
|
|
why = append(why, "dead")
|
|
}
|
|
if s.OOMKilled {
|
|
why = append(why, "killed for memory")
|
|
}
|
|
if s.State == "exited" && s.ExitCode != 0 {
|
|
why = append(why, fmt.Sprintf("exited %d", s.ExitCode))
|
|
}
|
|
if s.Restarts >= 5 {
|
|
why = append(why, fmt.Sprintf("restarted %d times", s.Restarts))
|
|
}
|
|
if len(why) == 0 {
|
|
continue
|
|
}
|
|
out = append(out, map[string]any{"name": s.Name, "state": s.State, "health": s.Health, "why": why,
|
|
"mesh_held": s.MeshHeld, "held_by": s.HeldBy, "image": s.Image, "finished_at": s.FinishedAt})
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// Ports is every port the containers publish on the machine.
|
|
func (c *Client) Ports(ctx context.Context) ([]map[string]any, error) {
|
|
all, err := c.Containers(ctx, "", "", "")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := []map[string]any{}
|
|
for _, s := range all {
|
|
for _, p := range s.Ports {
|
|
out = append(out, map[string]any{"published": p, "container": s.Name, "state": s.State, "mesh_held": s.MeshHeld})
|
|
}
|
|
if s.Network == "host" && s.State == "running" {
|
|
out = append(out, map[string]any{"published": "host network: every port it listens on", "container": s.Name, "state": s.State, "mesh_held": s.MeshHeld})
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// Unlabelled is every container the mesh does not hold: the cleanup list, with what each is
|
|
// likely to be (a compose project, a one-off).
|
|
func (c *Client) Unlabelled(ctx context.Context) (map[string]any, error) {
|
|
all, err := c.Containers(ctx, "other", "", "")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
running := 0
|
|
for _, s := range all {
|
|
if s.State == "running" {
|
|
running++
|
|
}
|
|
}
|
|
return map[string]any{"count": len(all), "running": running, "containers": all,
|
|
"note": "the mesh holds none of these; it never removes them. docker_prune with containers=true removes the stopped ones"}, nil
|
|
}
|
|
|
|
// ---- images ----------------------------------------------------------------------------------
|
|
|
|
// Image is one image with what uses it.
|
|
type Image struct {
|
|
ID string `json:"id"`
|
|
Repository string `json:"repository"`
|
|
Tag string `json:"tag"`
|
|
Created string `json:"created"`
|
|
Size string `json:"size"`
|
|
SizeBytes int64 `json:"size_bytes"`
|
|
Dangling bool `json:"dangling"`
|
|
UsedBy []string `json:"used_by"`
|
|
MeshUsed bool `json:"used_by_mesh"`
|
|
}
|
|
|
|
// Images is the images on the machine, the largest first, with the containers using each.
|
|
func (c *Client) Images(ctx context.Context, filter, match string, limit int) (map[string]any, error) {
|
|
out, err := c.docker(ctx, "image", "ls", "--no-trunc", "--format", "{{json .}}")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
raw, err := jsonLines[map[string]string](out)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
containers, err := c.inspectAll(ctx)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
users, meshUsers := map[string][]string{}, map[string]bool{}
|
|
for _, i := range containers {
|
|
s := summary(i)
|
|
users[i.Image] = append(users[i.Image], s.Name)
|
|
meshUsers[i.Image] = meshUsers[i.Image] || s.MeshHeld
|
|
}
|
|
images := []Image{}
|
|
var total int64
|
|
for _, r := range raw {
|
|
img := Image{ID: r["ID"], Repository: r["Repository"], Tag: r["Tag"], Created: r["CreatedAt"], Size: r["Size"],
|
|
SizeBytes: Bytes(r["Size"]), Dangling: r["Repository"] == "<none>" && r["Tag"] == "<none>"}
|
|
img.UsedBy = append([]string{}, users[img.ID]...)
|
|
img.MeshUsed = meshUsers[img.ID]
|
|
switch filter {
|
|
case "", "all":
|
|
case "dangling":
|
|
if !img.Dangling {
|
|
continue
|
|
}
|
|
case "unused":
|
|
if len(img.UsedBy) > 0 {
|
|
continue
|
|
}
|
|
case "used":
|
|
if len(img.UsedBy) == 0 {
|
|
continue
|
|
}
|
|
default:
|
|
return nil, fmt.Errorf("filter %q: all, dangling, unused or used", filter)
|
|
}
|
|
if match != "" && !strings.Contains(img.Repository+":"+img.Tag, match) {
|
|
continue
|
|
}
|
|
total += max(img.SizeBytes, 0)
|
|
images = append(images, img)
|
|
}
|
|
sort.Slice(images, func(a, b int) bool { return images[a].SizeBytes > images[b].SizeBytes })
|
|
count := len(images)
|
|
if len(images) > limit {
|
|
images = images[:limit]
|
|
}
|
|
return map[string]any{"count": count, "shown": len(images), "size_bytes_summed": total,
|
|
"note": "sizes are summed per image; images share layers, so the disk they take together is docker_disk_usage's", "images": images}, nil
|
|
}
|
|
|
|
// ---- prune -----------------------------------------------------------------------------------
|
|
|
|
// PruneAsk is what a prune is asked to take. Volumes are never among them: a volume is data.
|
|
type PruneAsk struct {
|
|
Images bool
|
|
BuildCache bool
|
|
Containers bool
|
|
OlderThanH int
|
|
DryRun bool
|
|
}
|
|
|
|
// Prune removes, or with DryRun only lists, dangling images, unused build cache and — only when
|
|
// asked — stopped containers the mesh does not hold. Never a volume, never a container the mesh
|
|
// holds, never an image a container uses (the daemon refuses that itself).
|
|
func (c *Client) Prune(ctx context.Context, a PruneAsk) (map[string]any, error) {
|
|
answer := map[string]any{"dry_run": a.DryRun, "volumes": "never pruned: a volume is data"}
|
|
until := []string{}
|
|
if a.OlderThanH > 0 {
|
|
until = []string{"--filter", fmt.Sprintf("until=%dh", a.OlderThanH)}
|
|
answer["older_than_hours"] = a.OlderThanH
|
|
}
|
|
if a.Images {
|
|
args := append([]string{"image", "ls", "--no-trunc", "--filter", "dangling=true", "--format", "{{json .}}"}, until...)
|
|
out, err := c.docker(ctx, args...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
raw, err := jsonLines[map[string]string](out)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var sum int64
|
|
ids := []string{}
|
|
for _, r := range raw {
|
|
sum += max(Bytes(r["Size"]), 0)
|
|
ids = append(ids, shortID(r["ID"]))
|
|
}
|
|
if len(ids) > 50 {
|
|
ids = ids[:50]
|
|
}
|
|
section := map[string]any{"dangling": len(raw), "size_bytes_summed": sum, "ids_first_50": ids}
|
|
if !a.DryRun && len(raw) > 0 {
|
|
out, err := c.docker(ctx, append([]string{"image", "prune", "--force"}, until...)...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
section["reclaimed"] = reclaimed(out)
|
|
}
|
|
answer["images"] = section
|
|
}
|
|
if a.BuildCache {
|
|
section := map[string]any{}
|
|
if a.DryRun {
|
|
out, err := c.docker(ctx, "system", "df", "--format", "{{json .}}")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
rows, err := jsonLines[map[string]string](out)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
for _, r := range rows {
|
|
if r["Type"] == "Build Cache" {
|
|
section["entries"], section["size"], section["reclaimable"] = r["TotalCount"], r["Size"], r["Reclaimable"]
|
|
}
|
|
}
|
|
section["note"] = "an estimate: a real prune takes the cache nothing refers to, which can be less than reclaimable"
|
|
} else {
|
|
out, err := c.docker(ctx, append([]string{"builder", "prune", "--force"}, until...)...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
section["reclaimed"] = reclaimed(out)
|
|
}
|
|
answer["build_cache"] = section
|
|
}
|
|
if a.Containers {
|
|
all, err := c.Containers(ctx, "other", "", "")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
cutoff := c.Now().Add(-time.Duration(a.OlderThanH) * time.Hour)
|
|
stopped := []string{}
|
|
for _, s := range all {
|
|
if s.State != "exited" && s.State != "created" && s.State != "dead" {
|
|
continue
|
|
}
|
|
if a.OlderThanH > 0 {
|
|
when := s.FinishedAt
|
|
if when == "" {
|
|
when = s.Created
|
|
}
|
|
if t, err := time.Parse(time.RFC3339Nano, when); err == nil && t.After(cutoff) {
|
|
continue
|
|
}
|
|
}
|
|
stopped = append(stopped, s.Name)
|
|
}
|
|
section := map[string]any{"stopped_not_held": stopped}
|
|
if !a.DryRun && len(stopped) > 0 {
|
|
// By name, and without --volumes: what they mounted stays.
|
|
if _, err := c.docker(ctx, append([]string{"container", "rm"}, stopped...)...); err != nil {
|
|
return nil, err
|
|
}
|
|
section["removed"] = stopped
|
|
}
|
|
answer["containers"] = section
|
|
}
|
|
if a.DryRun {
|
|
answer["note"] = "nothing was removed: call again with dry_run false to prune"
|
|
}
|
|
return answer, nil
|
|
}
|
|
|
|
func reclaimed(out string) string {
|
|
for _, l := range lines(out) {
|
|
if strings.HasPrefix(l, "Total reclaimed space:") || strings.HasPrefix(l, "Total:") {
|
|
return strings.TrimSpace(l[strings.Index(l, ":")+1:])
|
|
}
|
|
}
|
|
return "0B"
|
|
}
|
|
|
|
// ---- disk, networks, volumes, events, daemon ---------------------------------------------------
|
|
|
|
// DiskUsage is `docker system df -v`, summed per kind, with the largest of each.
|
|
func (c *Client) DiskUsage(ctx context.Context, top int) (map[string]any, error) {
|
|
out, err := c.docker(ctx, "system", "df", "--format", "{{json .}}")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
rows, err := jsonLines[map[string]any](out)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
summaryRows := []map[string]any{}
|
|
for _, r := range rows {
|
|
summaryRows = append(summaryRows, map[string]any{"type": r["Type"], "total": r["TotalCount"], "active": r["Active"],
|
|
"size": r["Size"], "size_bytes": Bytes(fmt.Sprint(r["Size"])), "reclaimable": r["Reclaimable"]})
|
|
}
|
|
verbose, err := c.docker(ctx, "system", "df", "--verbose", "--format", "{{json .}}")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var v map[string][]map[string]any
|
|
if err := json.Unmarshal([]byte(verbose), &v); err != nil {
|
|
return nil, fmt.Errorf("docker system df -v answered something that is not JSON: %v", err)
|
|
}
|
|
largest := map[string]any{}
|
|
for kind, items := range v {
|
|
sort.Slice(items, func(a, b int) bool { return Bytes(fmt.Sprint(items[a]["Size"])) > Bytes(fmt.Sprint(items[b]["Size"])) })
|
|
if len(items) > top {
|
|
items = items[:top]
|
|
}
|
|
trimmed := []map[string]any{}
|
|
for _, it := range items {
|
|
t := map[string]any{"size": it["Size"]}
|
|
for _, k := range []string{"Repository", "Tag", "ID", "Names", "Name", "Containers", "Links", "Image", "State", "Description", "InUse", "LastUsedSince", "UniqueSize"} {
|
|
if val, ok := it[k]; ok && val != nil && val != "" {
|
|
t[strings.ToLower(k)] = val
|
|
}
|
|
}
|
|
trimmed = append(trimmed, t)
|
|
}
|
|
largest[kind] = trimmed
|
|
}
|
|
return map[string]any{"summary": summaryRows, "largest": largest}, nil
|
|
}
|
|
|
|
// Networks is every network with its addressing and the running containers on it.
|
|
func (c *Client) Networks(ctx context.Context) ([]map[string]any, error) {
|
|
ids, err := c.docker(ctx, "network", "ls", "--quiet", "--no-trunc")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if len(lines(ids)) == 0 {
|
|
return []map[string]any{}, nil
|
|
}
|
|
out, err := c.docker(ctx, append([]string{"network", "inspect"}, lines(ids)...)...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var got []struct {
|
|
Name, Driver, Scope string
|
|
Internal bool
|
|
IPAM struct {
|
|
Config []struct{ Subnet, Gateway string }
|
|
}
|
|
Containers map[string]struct{ Name, IPv4Address string }
|
|
Labels map[string]string
|
|
}
|
|
if err := json.Unmarshal([]byte(out), &got); err != nil {
|
|
return nil, fmt.Errorf("docker network inspect answered something that is not JSON: %v", err)
|
|
}
|
|
held := map[string]bool{}
|
|
if all, err := c.Containers(ctx, "", "", ""); err == nil {
|
|
for _, s := range all {
|
|
held[s.Name] = s.MeshHeld
|
|
}
|
|
}
|
|
res := []map[string]any{}
|
|
for _, n := range got {
|
|
subnets := []string{}
|
|
for _, cfg := range n.IPAM.Config {
|
|
subnets = append(subnets, strings.TrimSpace(cfg.Subnet+" gw "+cfg.Gateway))
|
|
}
|
|
members := []map[string]any{}
|
|
for _, m := range n.Containers {
|
|
members = append(members, map[string]any{"name": m.Name, "address": m.IPv4Address, "mesh_held": held[m.Name]})
|
|
}
|
|
sort.Slice(members, func(a, b int) bool { return fmt.Sprint(members[a]["name"]) < fmt.Sprint(members[b]["name"]) })
|
|
res = append(res, map[string]any{"name": n.Name, "driver": n.Driver, "scope": n.Scope, "internal": n.Internal,
|
|
"subnets": subnets, "containers": members, "compose_project": n.Labels["com.docker.compose.project"]})
|
|
}
|
|
sort.Slice(res, func(a, b int) bool { return fmt.Sprint(res[a]["name"]) < fmt.Sprint(res[b]["name"]) })
|
|
return res, nil
|
|
}
|
|
|
|
// Volumes is every volume with the containers mounting it, whether the mesh holds any of them, and
|
|
// — when asked, which is slower — its size.
|
|
func (c *Client) Volumes(ctx context.Context, unmountedOnly, sizes bool) (map[string]any, error) {
|
|
names, err := c.docker(ctx, "volume", "ls", "--quiet")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
containers, err := c.Containers(ctx, "", "", "")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
mountedBy := map[string][]map[string]any{}
|
|
for _, s := range containers {
|
|
for _, m := range s.Mounts {
|
|
if m.Type == "volume" {
|
|
mountedBy[m.Name] = append(mountedBy[m.Name], map[string]any{"container": s.Name, "state": s.State, "mesh_held": s.MeshHeld})
|
|
}
|
|
}
|
|
}
|
|
size := map[string]string{}
|
|
if sizes {
|
|
out, err := c.docker(ctx, "system", "df", "--verbose", "--format", "{{json .}}")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var v struct{ Volumes []map[string]any }
|
|
if err := json.Unmarshal([]byte(out), &v); err == nil {
|
|
for _, vol := range v.Volumes {
|
|
size[fmt.Sprint(vol["Name"])] = fmt.Sprint(vol["Size"])
|
|
}
|
|
}
|
|
}
|
|
vols := []map[string]any{}
|
|
if len(lines(names)) > 0 {
|
|
out, err := c.docker(ctx, append([]string{"volume", "inspect"}, lines(names)...)...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var got []struct {
|
|
Name, Driver, CreatedAt, Mountpoint string
|
|
Labels map[string]string
|
|
}
|
|
if err := json.Unmarshal([]byte(out), &got); err != nil {
|
|
return nil, fmt.Errorf("docker volume inspect answered something that is not JSON: %v", err)
|
|
}
|
|
for _, v := range got {
|
|
by := mountedBy[v.Name]
|
|
if unmountedOnly && len(by) > 0 {
|
|
continue
|
|
}
|
|
mesh := false
|
|
for _, b := range by {
|
|
mesh = mesh || b["mesh_held"].(bool)
|
|
}
|
|
_, anonymous := v.Labels["com.docker.volume.anonymous"]
|
|
entry := map[string]any{"name": v.Name, "driver": v.Driver, "created": v.CreatedAt, "anonymous": anonymous,
|
|
"compose_project": v.Labels["com.docker.compose.project"], "mounted_by": append([]map[string]any{}, by...), "mesh_held": mesh}
|
|
if sizes {
|
|
entry["size"] = size[v.Name]
|
|
}
|
|
vols = append(vols, entry)
|
|
}
|
|
}
|
|
return map[string]any{"count": len(vols), "volumes": vols,
|
|
"note": "mesh_held: a container the mesh holds mounts it. Nothing here removes a volume; docker_prune never does"}, nil
|
|
}
|
|
|
|
// Events is what the runtime did in a window ending now, the latest last.
|
|
func (c *Client) Events(ctx context.Context, minutes int, kind string, limit int, execs bool) (map[string]any, error) {
|
|
args := []string{"events", "--since", fmt.Sprintf("%dm", minutes), "--until", "0s", "--format", "{{json .}}"}
|
|
if kind != "" {
|
|
switch kind {
|
|
case "container", "image", "network", "volume", "daemon", "plugin", "builder":
|
|
default:
|
|
return nil, fmt.Errorf("type %q: container, image, network, volume, daemon, plugin or builder", kind)
|
|
}
|
|
args = append(args, "--filter", "type="+kind)
|
|
}
|
|
out, err := c.docker(ctx, args...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
raw, err := jsonLines[struct {
|
|
Type, Action string
|
|
Actor struct {
|
|
ID string
|
|
Attributes map[string]string
|
|
}
|
|
TimeNano int64 `json:"timeNano"`
|
|
}](out)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
events := []map[string]any{}
|
|
for _, e := range raw {
|
|
if !execs && strings.HasPrefix(e.Action, "exec_") {
|
|
continue
|
|
}
|
|
_, held := e.Actor.Attributes[MeshLabel]
|
|
ev := map[string]any{"time": time.Unix(0, e.TimeNano).UTC().Format(time.RFC3339), "type": e.Type, "action": e.Action,
|
|
"id": shortID(e.Actor.ID), "name": e.Actor.Attributes["name"]}
|
|
if e.Type == "container" {
|
|
ev["mesh_held"] = held
|
|
ev["image"] = e.Actor.Attributes["image"]
|
|
if code, ok := e.Actor.Attributes["exitCode"]; ok {
|
|
ev["exit_code"] = code
|
|
}
|
|
}
|
|
events = append(events, ev)
|
|
}
|
|
total := len(events)
|
|
if len(events) > limit {
|
|
events = events[len(events)-limit:]
|
|
}
|
|
return map[string]any{"minutes": minutes, "count": total, "shown": len(events), "events": events}, nil
|
|
}
|
|
|
|
// restartOnly are the daemon keys the runtime reads only when it starts: a reload leaves them as
|
|
// they were (dockerd's reloadable set is debug, labels, live-restore, registries, max-concurrent-*,
|
|
// default-runtime, runtimes, shutdown-timeout and a few more).
|
|
var restartOnly = []string{"dns", "dns-opts", "dns-search", "data-root", "storage-driver", "log-driver", "log-opts", "bip",
|
|
"default-address-pools", "iptables", "ip6tables", "ipv6", "userns-remap", "exec-opts", "allow-direct-routing", "mtu"}
|
|
|
|
// DaemonConfig is the runtime's file and what the daemon runs with now, and where the two differ.
|
|
func (c *Client) DaemonConfig(ctx context.Context) (map[string]any, error) {
|
|
answer := map[string]any{"file": DaemonFile}
|
|
var file map[string]any
|
|
raw, err := c.ReadFile(DaemonFile)
|
|
switch {
|
|
case os.IsNotExist(err):
|
|
answer["file_state"] = "absent: the daemon runs on its defaults"
|
|
case err != nil:
|
|
answer["file_state"] = "unreadable: " + err.Error()
|
|
default:
|
|
if err := json.Unmarshal(raw, &file); err != nil {
|
|
answer["file_state"] = "not JSON — the daemon refuses to start with it: " + err.Error()
|
|
} else {
|
|
answer["file_state"] = "read"
|
|
answer["keys"] = file
|
|
}
|
|
}
|
|
out, err := c.docker(ctx, "info", "--format", "{{json .}}")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var info map[string]any
|
|
if err := json.Unmarshal([]byte(out), &info); err != nil {
|
|
return nil, fmt.Errorf("docker info answered something that is not JSON: %v", err)
|
|
}
|
|
essentials := map[string]any{}
|
|
for _, k := range []string{"ServerVersion", "Driver", "LoggingDriver", "CgroupDriver", "CgroupVersion", "LiveRestoreEnabled",
|
|
"DockerRootDir", "Containers", "ContainersRunning", "ContainersPaused", "ContainersStopped", "Images", "KernelVersion",
|
|
"OperatingSystem", "NCPU", "MemTotal", "DefaultRuntime", "FirewallBackend", "SecurityOptions", "Warnings", "Debug"} {
|
|
if v, ok := info[k]; ok {
|
|
essentials[k] = v
|
|
}
|
|
}
|
|
if reg, ok := info["RegistryConfig"].(map[string]any); ok {
|
|
insecure := []string{}
|
|
if idx, ok := reg["IndexConfigs"].(map[string]any); ok {
|
|
for name, cfg := range idx {
|
|
if m, ok := cfg.(map[string]any); ok && m["Secure"] == false {
|
|
insecure = append(insecure, name)
|
|
}
|
|
}
|
|
}
|
|
sort.Strings(insecure)
|
|
essentials["InsecureRegistries"] = insecure
|
|
essentials["InsecureRegistryCIDRs"] = reg["InsecureRegistryCIDRs"]
|
|
}
|
|
answer["daemon"] = essentials
|
|
|
|
// Where the file says one thing and the daemon runs another: a key changed since the last reload,
|
|
// or one only a restart takes.
|
|
pending := []string{}
|
|
if file != nil {
|
|
if v, ok := file["live-restore"].(bool); ok && info["LiveRestoreEnabled"] != v {
|
|
pending = append(pending, fmt.Sprintf("live-restore is %v in the file and %v in the daemon: a reload takes it", v, info["LiveRestoreEnabled"]))
|
|
}
|
|
if v, ok := file["log-driver"].(string); ok && info["LoggingDriver"] != v {
|
|
pending = append(pending, fmt.Sprintf("log-driver is %s in the file and %v in the daemon: only a restart takes it", v, info["LoggingDriver"]))
|
|
}
|
|
present := []string{}
|
|
for _, k := range restartOnly {
|
|
if _, ok := file[k]; ok {
|
|
present = append(present, k)
|
|
}
|
|
}
|
|
answer["read_only_at_start"] = present
|
|
}
|
|
answer["pending"] = pending
|
|
answer["note"] = "keys read only at start take effect at the daemon's next restart; with live-restore on, a restart keeps every container running"
|
|
return answer, nil
|
|
}
|