- A reopened fault keeps its gaps: it was there at a send unless the send fell in one, so a send that breaks a machine recovered before it still fails its gate (A2). - An undecided part holds only the conditions that name it (A4). - D2 holds a silent resolver for the next run again, refused or not: a burst of refusals is also a restart (A3). - A late merge is planned at the newest planned merge of its branch in any state, not only an open one (A1); merge times to the nanosecond (A5).
222 lines
10 KiB
Go
222 lines
10 KiB
Go
package main
|
|
|
|
import (
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/novox/mesh-controller/internal/catalogue"
|
|
"github.com/novox/mesh-controller/internal/conditions"
|
|
"github.com/novox/mesh-controller/internal/inventory"
|
|
"github.com/novox/mesh-controller/internal/lease"
|
|
"github.com/novox/mesh-controller/internal/link"
|
|
)
|
|
|
|
// novox/hq issue 348: on 2026-10-09 the control node's resolver stopped answering on its private address
|
|
// at 10:57:57 UTC; the machine's network condition was raised at 10:58:45. The node-engine and the
|
|
// controller were sent at 10:59:34. The new node-engine's first statement judged the names once — unknown,
|
|
// "one look failed; a second decides" — and that statement cleared the condition, though its last evidence
|
|
// still said "connection refused". The next look raised it again at 11:00:23, after the send, and both
|
|
// builds failed their gate at 11:10 with "raised since it was sent" and were put back, for a fault that
|
|
// began before they were sent.
|
|
|
|
// namesRefused is the control node's names part as its node-engine said it in the outage: its own
|
|
// resolver, at its own address, refusing.
|
|
func namesRefused(state string, streak int) link.NetworkPart {
|
|
p := link.NetworkPart{Part: link.PartNames, State: state, Since: h0, Streak: streak}
|
|
if state == link.StateUnhealthy {
|
|
p.Reason = "1 of its 2 resolvers do not answer as the mesh's do"
|
|
p.Said = "10.77.0.1 — anchor.internal (IPv4): read udp 10.77.0.1:35244->10.77.0.1:53: read: connection refused"
|
|
p.Toward = []string{"10.77.0.1"}
|
|
}
|
|
return p
|
|
}
|
|
|
|
func networkSaying(parts ...link.NetworkPart) *link.NetworkHealth {
|
|
state := link.StateHealthy
|
|
for _, p := range parts {
|
|
switch {
|
|
case p.State == link.StateUnhealthy:
|
|
state = link.StateUnhealthy
|
|
case p.State == link.StateUnknown && state == link.StateHealthy:
|
|
state = link.StateUnknown
|
|
}
|
|
}
|
|
return &link.NetworkHealth{State: state, Since: h0, Parts: parts}
|
|
}
|
|
|
|
// TestReplay348 replays the statements of the outage: the condition raised before the send is not
|
|
// cleared by the restarted engine's first, undecided statement, and the gate does not count it against
|
|
// the builds sent after it began.
|
|
func TestReplay348(t *testing.T) {
|
|
open := aMesh(t)
|
|
ctx := t.Context()
|
|
inv, k := open.inventory, conditionsFrom
|
|
say := func(at time.Time, n *link.NetworkHealth) {
|
|
t.Helper()
|
|
if err := stateHealth(ctx, inv, k, "anchor", link.Health{Contract: link.ReadinessContract, At: at, Network: n}, at); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
}
|
|
network := func() (conditions.Condition, bool) {
|
|
t.Helper()
|
|
list, err := k.Open(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
for _, c := range list {
|
|
if c.Key == "machine.anchor.network" {
|
|
return c, true
|
|
}
|
|
}
|
|
return conditions.Condition{}, false
|
|
}
|
|
|
|
// 10:58:45 — the second failing look: raised.
|
|
say(h0, networkSaying(namesRefused(link.StateUnhealthy, 2)))
|
|
raised, ok := network()
|
|
if !ok {
|
|
t.Fatal("the resolver refusing on the control node raised nothing")
|
|
}
|
|
time.Sleep(5 * time.Millisecond)
|
|
sent := time.Now().UTC()
|
|
time.Sleep(5 * time.Millisecond)
|
|
|
|
// 10:59:42 — the restarted engine's first statement: one look failed, a second decides.
|
|
say(h0.Add(time.Minute), networkSaying(namesRefused(link.StateUnknown, 1)))
|
|
if _, ok := network(); !ok {
|
|
t.Fatal("a statement that judged nothing yet cleared the condition: the restarted engine's first look " +
|
|
"said the fault was gone while it still refused")
|
|
}
|
|
// 11:00:23 — its second look: unhealthy again, the same raising.
|
|
say(h0.Add(2*time.Minute), networkSaying(namesRefused(link.StateUnhealthy, 2)))
|
|
again, ok := network()
|
|
if !ok || !again.Raised.Equal(raised.Raised) || again.Count != 1 {
|
|
t.Fatalf("the same raising was not kept: raised %s (first %s), count %d", again.Raised, raised.Raised, again.Count)
|
|
}
|
|
|
|
// The gate on the control node, for a build sent after the fault began.
|
|
open2, err := k.Open(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
f := gateFacts{judged: true, open: open2}
|
|
if w := aboutTheMachine("anchor", []string{"mesh-host"}, sent, f); w.whole != "" || len(w.on) != 0 {
|
|
t.Fatalf("a fault from before the send held the build: %+v", w)
|
|
}
|
|
|
|
// Decided healthy: cleared.
|
|
say(h0.Add(3*time.Minute), networkSaying(link.NetworkPart{Part: link.PartNames, State: link.StateHealthy, Since: h0}))
|
|
if c, ok := network(); ok {
|
|
t.Fatalf("a statement that decides the names healthy left %s open", c.Key)
|
|
}
|
|
}
|
|
|
|
// A fault there at the send, cleared and reopened after it, is not raised since the send; one that cleared
|
|
// before the send and came back after it is — a send that breaks a recovered machine fails its gate (review
|
|
// of mesh-controller PR 179, A2). Read through OpenAt, by both of the gate's readings.
|
|
func TestAFaultThatFlappedAfterTheSendIsNotTheSendsAndOneThatRecoveredBeforeItIs(t *testing.T) {
|
|
since := h0
|
|
network := func(first time.Time, gaps ...conditions.Gap) conditions.Condition {
|
|
raised := since.Add(time.Minute)
|
|
if len(gaps) > 0 {
|
|
raised = gaps[len(gaps)-1].Reopened
|
|
}
|
|
return conditions.Condition{Key: "machine.anchor.network", Kind: kindMachineNetwork,
|
|
Subject: conditions.Subject{Scope: conditions.ScopeMachine, ID: "anchor", Machine: "anchor"},
|
|
Summary: "anchor's network is not healthy", Source: sourceNetwork, First: first, Gaps: gaps, Raised: raised}
|
|
}
|
|
held := func(c conditions.Condition) bool {
|
|
return aboutTheMachine("anchor", []string{"mesh-controller"}, since, gateFacts{judged: true,
|
|
open: []conditions.Condition{c}}).whole != ""
|
|
}
|
|
// The day's case: raised before the send, cleared 8 s after it, reopened 49 s after it.
|
|
flapped := network(since.Add(-49*time.Second),
|
|
conditions.Gap{Cleared: since.Add(8 * time.Second), Reopened: since.Add(49 * time.Second)})
|
|
if held(flapped) {
|
|
t.Fatal("a fault there at the send, flapping after it, held the machine")
|
|
}
|
|
// Recovered before the send, broken again after it: the send's.
|
|
recovered := network(since.Add(-time.Hour),
|
|
conditions.Gap{Cleared: since.Add(-30 * time.Second), Reopened: since.Add(20 * time.Second)})
|
|
if !held(recovered) {
|
|
t.Fatal("a machine recovered at the send and broken after it passed the gate")
|
|
}
|
|
// An older gap, before the send, and the fault there at the send: not the send's.
|
|
twice := network(since.Add(-time.Hour),
|
|
conditions.Gap{Cleared: since.Add(-50 * time.Minute), Reopened: since.Add(-45 * time.Minute)},
|
|
conditions.Gap{Cleared: since.Add(10 * time.Second), Reopened: since.Add(30 * time.Second)})
|
|
if held(twice) {
|
|
t.Fatal("a fault there at the send, with an older gap, held the machine")
|
|
}
|
|
// Raised after the send, never cleared: the send's.
|
|
if !held(network(time.Time{})) {
|
|
t.Fatal("a fault raised after the send held nothing")
|
|
}
|
|
// And a module's own, through judgeHealth.
|
|
at := since.Add(2 * time.Minute)
|
|
g := gateFacts{judged: true, now: at, reports: map[string]inventory.Reported{"anchor": {Node: "anchor",
|
|
Outcome: inventory.OutcomeApplied, At: &at, Current: true}}, engines: map[string]string{},
|
|
served: map[string]served{}, rolledBack: map[string][]lease.Rollback{},
|
|
open: []conditions.Condition{{Key: "provider.app.anchor.x.failing", Subject: conditions.Subject{
|
|
Scope: conditions.ScopeProvider, ID: "app.anchor.x", Machine: "anchor"}, Summary: "failing",
|
|
First: since.Add(-time.Hour), Raised: since.Add(time.Minute),
|
|
Gaps: []conditions.Gap{{Cleared: since.Add(5 * time.Second), Reopened: since.Add(time.Minute)}}}}}
|
|
if _, why := judgeHealth("app", "", catalogue.Manifest{Module: "app"}, "anchor", since, g); strings.HasPrefix(why, "raised since it was sent") {
|
|
t.Fatalf("a module's own fault there at the send: %s", why)
|
|
}
|
|
g.open[0].Gaps[0].Cleared = since.Add(-5 * time.Second)
|
|
if _, why := judgeHealth("app", "", catalogue.Manifest{Module: "app"}, "anchor", since, g); !strings.HasPrefix(why, "raised since it was sent") {
|
|
t.Fatalf("a module's own fault, recovered at the send and back after it, was not counted: %s", why)
|
|
}
|
|
}
|
|
|
|
// Only an undecided part holds a condition that names it; a condition about another part clears, and a
|
|
// statement unknown as a whole holds every part (review of PR 179, A4). Pure.
|
|
func TestAnUndecidedPartHoldsOnlyWhatNamesIt(t *testing.T) {
|
|
f := netFacts(map[string]*inventory.NetworkHealth{
|
|
"anchor": aNetwork(link.StateUnknown, inventory.NetworkPart{Part: link.PartNames, State: link.StateUnknown, Streak: 1},
|
|
inventory.NetworkPart{Part: link.PartRoute, State: link.StateHealthy}),
|
|
"laptop": aNetwork(link.StateHealthy, inventory.NetworkPart{Part: link.PartNames, State: link.StateHealthy}),
|
|
"spare": aNetwork(link.StateStarting, inventory.NetworkPart{Part: link.PartTunnel, State: link.StateHealthy}),
|
|
"other": aNetwork(link.StateStarting, inventory.NetworkPart{Part: link.PartTunnel, State: link.StateStarting}),
|
|
})
|
|
u := undecidedParts(f)
|
|
if !u["anchor"][link.PartNames] || u["anchor"][link.PartRoute] || u["laptop"] != nil || !u["spare"]["*"] ||
|
|
!u["other"][link.PartTunnel] || u["other"]["*"] {
|
|
t.Fatalf("undecided: %v", u)
|
|
}
|
|
about := func(machine, said string, also ...string) conditions.Condition {
|
|
return conditions.Condition{Subject: conditions.Subject{Scope: conditions.ScopeMachine, ID: machine,
|
|
Machine: machine, Also: also}, Evidence: []conditions.Evidence{{Said: said}}}
|
|
}
|
|
for _, c := range []struct {
|
|
c conditions.Condition
|
|
held bool
|
|
}{
|
|
{about("anchor", "names since 2026-10-09 10:58:45 UTC: 10.77.0.1 — refused"), true},
|
|
{about("anchor", "route since 2026-10-09 10:58:45 UTC: no default route"), false},
|
|
{about("laptop", "names since 2026-10-09 10:58:45 UTC: refused"), false},
|
|
{about("spare", "route since …: no default route"), true},
|
|
{about("hub", "anchor: names: refused", "anchor"), true},
|
|
{about("hub", "anchor: tunnel: no handshake", "anchor"), false},
|
|
} {
|
|
if got := heldUndecided(c.c, u); got != c.held {
|
|
t.Errorf("%s %q held %v, want %v", c.c.Subject.Machine, c.c.Evidence[0].Said, got, c.held)
|
|
}
|
|
}
|
|
}
|
|
|
|
// A release walks its modules without a record per module: D10 counts what its tier names as rolling,
|
|
// so the node-engine a release walks is not "behind, and no plan is rolling it out" on its first machine.
|
|
func TestAReleaseRollsOutWhatItsTierNames(t *testing.T) {
|
|
plans := []inventory.Plan{
|
|
{ID: "release-1", State: inventory.PlanRolling, Tiers: [][]string{{"mesh-host"}}, Modules: map[string]*inventory.PlanModule{}},
|
|
{ID: "plan-2", State: inventory.PlanRolling, Modules: map[string]*inventory.PlanModule{"letta": {}}},
|
|
}
|
|
got := rollingModules(plans)
|
|
if !got["mesh-host"] || !got["letta"] || len(got) != 2 {
|
|
t.Fatalf("rolling: %v", got)
|
|
}
|
|
}
|