Step 5 of the connectivity order. Every node's internal name resolves to its overlay address, on every node, computed centrally because it needs every node at once. Under `.internal`, which IANA reserved for exactly this in 2024 -- a name there can never collide with a public one, so an internal name that leaks into a public resolver fails rather than reaching a stranger's machine. The suffix is settable for a mesh that wants its own. Delivered in the same declaration as the peer list rather than a second one. A node holding the peers and not the names, or the reverse, is half on the network for as long as that lasts. This is not the /etc/hosts floor the design removes. That floor existed because a node had to reach the mesh's database before its own DNS worked -- a fallback for a circularity that is now gone. This is the mechanism: the complete set of names, generated whole and owned by the mesh, rather than a patch written underneath something else. A resolver daemon becomes necessary when names are wanted that are not one-per-node, and that is not yet true. A node resolves its own name to its overlay address rather than a loopback, because a service binding to the name it was given would otherwise listen somewhere nothing else can reach -- and the failure would appear on every other machine rather than that one. A node with no address gets no name. A name resolving to nothing is worse than no name: connecting to an address that does not answer hangs, where a name that does not resolve fails at once and says which name it was. Found while writing it: a test asserting every file in the declaration is mode 0600 would have forced /etc/hosts to 0600 and broken every lookup on the machine, to protect a file that is not secret. Verified in the lab: three machines, nine name lookups, each resolving to the right overlay address and reaching it.
179 lines
7.6 KiB
Go
179 lines
7.6 KiB
Go
package overlay
|
|
|
|
import (
|
|
"encoding/json"
|
|
"fmt"
|
|
"strings"
|
|
)
|
|
|
|
// The overlay is delivered as an ordinary declaration.
|
|
//
|
|
// novox/hq 08-connectivity: what the host receives is an interface configuration and a peer list,
|
|
// as files. It does not compute them and it does not know what a private network is — the shapes
|
|
// it already has are enough, which is why connectivity needs nothing new from tier 0.
|
|
//
|
|
// **No private key travels.** The configuration points at a file the node wrote from a key the
|
|
// mesh has never seen, using WireGuard's own ability to set one after the interface is up. So the
|
|
// control plane composes a complete configuration for a node it cannot impersonate.
|
|
|
|
// Interface is what the private network is called on a machine, and where the node's own key
|
|
// lives. A constant rather than a setting: two nodes disagreeing about the name would produce a
|
|
// mesh where each is configured correctly and nothing meets.
|
|
const (
|
|
Interface = "mesh0"
|
|
ConfigPath = "/etc/wireguard/" + Interface + ".conf"
|
|
Unit = "wg-quick@" + Interface
|
|
DefaultKeyPath = "/var/lib/mesh-host/overlay.key"
|
|
)
|
|
|
|
// Resource is one entry in a declaration, built here and read by the host.
|
|
type Resource map[string]any
|
|
|
|
// Declaration is what the mesh sends a node to put it on the network.
|
|
//
|
|
// Three resources and nothing clever: the tools, the configuration, and the interface running. A
|
|
// person can read it, which is the point — this is the first thing a node is ever told, and if it
|
|
// is wrong the node is unreachable and the mistake has to be findable by eye.
|
|
func Declaration(node Node, peers []Peer, everyone []Node, keyPath string) ([]byte, error) {
|
|
if node.Address == "" {
|
|
return nil, fmt.Errorf("%s has no address on the overlay, so there is nothing to configure",
|
|
node.Name)
|
|
}
|
|
if keyPath == "" {
|
|
keyPath = DefaultKeyPath
|
|
}
|
|
|
|
resources := []Resource{
|
|
{
|
|
"id": "overlay-tools", "type": "package", "package": "wireguard-tools",
|
|
},
|
|
{
|
|
"id": "overlay-config", "type": "file", "path": ConfigPath,
|
|
// Readable only by root: it lists every peer's key and endpoint, which is a map of
|
|
// the mesh. Not secret in the way a private key is, and not something to leave
|
|
// world-readable on a machine somebody else also uses.
|
|
"mode": "0600",
|
|
"content": config(node, peers, keyPath),
|
|
},
|
|
{
|
|
"id": "overlay-up", "type": "service", "unit": Unit,
|
|
"state": "running",
|
|
// Enabled, so the node comes back onto the network after a reboot without waiting to
|
|
// be told again. A node whose overlay only exists while something is watching is not
|
|
// a node that survives being switched off and on.
|
|
"boot": "enabled",
|
|
// And restarted when the peer list changes, because a running interface does not
|
|
// re-read its configuration.
|
|
//
|
|
// This is the whole of it: a node joins, every existing node's peer list changes,
|
|
// each file is replaced — and without this the service is already running, nothing
|
|
// reloads it, and every node keeps a network that no longer matches the mesh. It
|
|
// reports complete success. The lab found it the moment a third node arrived.
|
|
//
|
|
// Declared state rather than a command. The service must reflect the file; the host
|
|
// works out that it does not. A command to restart would be an action, and the link
|
|
// may not carry one (novox/hq ADR 0005) — the host refused exactly that, correctly,
|
|
// which is how this shape was arrived at.
|
|
"restart-on": []string{"overlay-config"},
|
|
},
|
|
}
|
|
|
|
// And the names, which come from the same graph and arrive in the same declaration. Separate
|
|
// steps in the design and one delivery in practice: a node that had the peers and not the
|
|
// names, or the reverse, would be half on the network for as long as that lasted.
|
|
names, err := Hosts(everyone, node.Name)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
resources = append(resources, Resource{
|
|
"id": "mesh-names", "type": "file", "path": HostsPath,
|
|
"mode": "0644", "content": names,
|
|
})
|
|
|
|
return json.Marshal(map[string]any{"declaration": 1, "resources": resources})
|
|
}
|
|
|
|
// config writes the interface file.
|
|
//
|
|
// Deliberately in the order a person would read it: who I am, then who I talk to, each with a
|
|
// line saying why it is there. A generated file that cannot be understood by the person it
|
|
// confuses is a generated file that gets edited by hand.
|
|
func config(node Node, peers []Peer, keyPath string) string {
|
|
var b strings.Builder
|
|
b.WriteString("# Generated by the mesh. Do not edit — this file is replaced whenever the\n")
|
|
b.WriteString("# peer graph changes, and an edit would survive until the next change and\n")
|
|
b.WriteString("# then vanish, which is worse than not being applied at all.\n")
|
|
fmt.Fprintf(&b, "#\n# node %s", node.Name)
|
|
if node.Site != "" {
|
|
fmt.Fprintf(&b, ", at %s", node.Site)
|
|
}
|
|
if !node.Reachable() {
|
|
b.WriteString(", not dialable — it opens every path itself")
|
|
}
|
|
b.WriteString("\n\n[Interface]\n")
|
|
fmt.Fprintf(&b, "Address = %s/32\n", node.Address)
|
|
if node.Reachable() {
|
|
if port := portOf(node.Endpoint); port != "" {
|
|
fmt.Fprintf(&b, "ListenPort = %s\n", port)
|
|
}
|
|
}
|
|
// The private key is set from a file the node wrote, so it never appears here and never
|
|
// travelled. Everything else in this file came from the mesh; this one line is the node's.
|
|
fmt.Fprintf(&b, "PostUp = wg set %%i private-key %s\n", keyPath)
|
|
|
|
if node.Hub {
|
|
// A hub carries traffic *between* its spokes, and a Linux machine does not forward
|
|
// packets unless it is told to. Without this every spoke reaches the hub perfectly and
|
|
// no spoke reaches any other — which is exactly how it failed in the lab, and it looks
|
|
// like a peering problem rather than a kernel setting.
|
|
//
|
|
// Here rather than in a separate resource because it is part of what being a hub means,
|
|
// and because it should last exactly as long as the interface does: a machine that stops
|
|
// being the hub should stop forwarding, and PostDown below is how that happens.
|
|
b.WriteString("PostUp = sysctl -q -w net.ipv4.ip_forward=1\n")
|
|
b.WriteString("PostDown = sysctl -q -w net.ipv4.ip_forward=0\n")
|
|
|
|
// And past the machine's own firewall, which on any node with a container runtime is
|
|
// closed. Docker sets the FORWARD policy to DROP and inserts its chains, so the substrate
|
|
// this mesh installs at tier 1 silently breaks the network it builds at tier 2: every
|
|
// spoke reaches the hub, no spoke reaches any other, and every part of it reports
|
|
// success. Found in the lab; nothing about it is visible from the mesh's own state.
|
|
//
|
|
// Inserted at the top so it precedes those chains, and removed on the way down so a
|
|
// machine that stops being the hub stops carrying other people's traffic. Guarded on
|
|
// iptables existing: a machine without it has no policy to get past.
|
|
for _, direction := range []string{"-i", "-o"} {
|
|
fmt.Fprintf(&b,
|
|
"PostUp = command -v iptables >/dev/null && iptables -I FORWARD 1 %s %%i -j ACCEPT || true\n",
|
|
direction)
|
|
}
|
|
for _, direction := range []string{"-i", "-o"} {
|
|
fmt.Fprintf(&b,
|
|
"PostDown = command -v iptables >/dev/null && iptables -D FORWARD %s %%i -j ACCEPT || true\n",
|
|
direction)
|
|
}
|
|
}
|
|
|
|
for _, p := range peers {
|
|
fmt.Fprintf(&b, "\n# %s — %s\n[Peer]\n", p.Name, p.Why)
|
|
fmt.Fprintf(&b, "PublicKey = %s\n", p.Key)
|
|
fmt.Fprintf(&b, "AllowedIPs = %s\n", p.Allowed)
|
|
if p.Endpoint != "" {
|
|
fmt.Fprintf(&b, "Endpoint = %s\n", p.Endpoint)
|
|
} else {
|
|
b.WriteString("# no endpoint: this peer cannot be dialled and opens the path itself\n")
|
|
}
|
|
if p.Keepalive {
|
|
b.WriteString("PersistentKeepalive = 25\n")
|
|
}
|
|
}
|
|
return b.String()
|
|
}
|
|
|
|
func portOf(endpoint string) string {
|
|
if i := strings.LastIndex(endpoint, ":"); i >= 0 {
|
|
return endpoint[i+1:]
|
|
}
|
|
return ""
|
|
}
|