The pipeline was observable from a merge to an artifact and went dark where it touched a machine: a node's report is control traffic only the control plane reads, so nothing said which version a machine runs, or that it refused to (novox/hq ADR 0134). The control plane now states both under the seat it holds — a role's events belong to the role and keep their address when the holder is replaced — and only when the report is news, because a machine reconciles every minute and a fact per report would be a fact per minute per machine. Whether a report is news is the store's answer: it holds the previous one, so the listener returns it and the server states the fact. That also gives the catch-up replay a subject the controller may publish: it was published as a module's event from a module called "control-plane", which does not exist, so the controller's own account refused it and every catalogue that asked what it missed was answered with nothing.
338 lines
11 KiB
Go
338 lines
11 KiB
Go
package inventory
|
|
|
|
import (
|
|
"context"
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// What each machine did with what it was last sent.
|
|
//
|
|
// Until this, a refusal or a failure moved last_seen and the reason went to a log line — so
|
|
// "which machine is not doing what it was told" had no answer the next morning, which is the
|
|
// question a mesh exists to answer.
|
|
|
|
func nodeNamed(t *testing.T, inv *Inventory, name string) string {
|
|
t.Helper()
|
|
n, err := inv.AddNode(context.Background(), name)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
return n.ID
|
|
}
|
|
|
|
func TestRefusedAndFailedAreDifferentSituations(t *testing.T) {
|
|
// Refused means the machine is exactly as it was, and what is wrong is in what was sent.
|
|
// Failed means it is in a state nobody declared, and what is wrong is on the machine. One
|
|
// word for both would make the report say less than the node did.
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
refuser := nodeNamed(t, inv, "refuser")
|
|
failer := nodeNamed(t, inv, "failer")
|
|
|
|
if _, err := inv.RecordDoing(ctx, refuser, Doing{
|
|
Outcome: OutcomeRefused, Refused: "resource \"x\": a file needs a path",
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if _, err := inv.RecordDoing(ctx, failer, Doing{
|
|
Outcome: OutcomeFailed,
|
|
Failed: []FailedResource{{ID: "svc", Error: "unit not found"}},
|
|
Applied: 4,
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 2 {
|
|
t.Fatalf("got %d", len(wrong))
|
|
}
|
|
by := map[string]Doing{}
|
|
for _, d := range wrong {
|
|
by[d.Node] = d
|
|
}
|
|
if by["refuser"].Outcome != OutcomeRefused || !strings.Contains(by["refuser"].Refused, "needs a path") {
|
|
t.Fatalf("got %+v", by["refuser"])
|
|
}
|
|
if by["failer"].Outcome != OutcomeFailed || len(by["failer"].Failed) != 1 {
|
|
t.Fatalf("got %+v", by["failer"])
|
|
}
|
|
// And how much DID work, because "four of five" and "none of five" are different machines.
|
|
if by["failer"].Applied != 4 {
|
|
t.Fatalf("what did apply was not kept: %+v", by["failer"])
|
|
}
|
|
}
|
|
|
|
func TestAMachineDoingWhatItWasToldIsNotOnTheList(t *testing.T) {
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "fine")
|
|
if _, err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeApplied, Applied: 6}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 0 {
|
|
t.Fatalf("a machine that did as it was told is listed as wrong: %+v", wrong)
|
|
}
|
|
// It is still answerable about, which is a different question.
|
|
d, said, err := inv.DoingOf(ctx, "fine")
|
|
if err != nil || !said {
|
|
t.Fatalf("said=%v err=%v", said, err)
|
|
}
|
|
if d.Wrong() || d.Applied != 6 {
|
|
t.Fatalf("got %+v", d)
|
|
}
|
|
}
|
|
|
|
func TestTheLastReportReplacesTheOneBefore(t *testing.T) {
|
|
// The question is the machine's CURRENT state. "This failed an hour ago and then succeeded"
|
|
// is not a machine anybody needs to look at, and a table of every report would bury the ones
|
|
// that matter under the ones that do not.
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "recovered")
|
|
if _, err := inv.RecordDoing(ctx, id, Doing{
|
|
Outcome: OutcomeFailed, Failed: []FailedResource{{ID: "a", Error: "no"}},
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if _, err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeApplied, Applied: 3}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 0 {
|
|
t.Fatalf("a machine that recovered is still listed as wrong: %+v", wrong)
|
|
}
|
|
d, _, _ := inv.DoingOf(ctx, "recovered")
|
|
if len(d.Failed) != 0 {
|
|
t.Fatalf("the old failure survived: %+v", d)
|
|
}
|
|
}
|
|
|
|
func TestAMachineThatHasSaidNothingIsNotWrong(t *testing.T) {
|
|
// It may be new, switched off, or unreachable. None of those is a machine that tried and
|
|
// could not, and treating silence as failure would have somebody debug a machine that has
|
|
// simply never been sent anything.
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
nodeNamed(t, inv, "silent")
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 0 {
|
|
t.Fatalf("silence was read as failure: %+v", wrong)
|
|
}
|
|
if _, said, err := inv.DoingOf(ctx, "silent"); err != nil || said {
|
|
t.Fatalf("a machine that never reported was reported about: said=%v err=%v", said, err)
|
|
}
|
|
}
|
|
|
|
func TestWhatANodeSaidGoesWhenTheNodeDoes(t *testing.T) {
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "leaving")
|
|
if _, err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeFailed}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if _, err := inv.store.Pool().Exec(ctx, `delete from node where name = 'leaving'`); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
var left int
|
|
if err := inv.store.Pool().QueryRow(ctx, `select count(*) from node_report`).Scan(&left); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if left != 0 {
|
|
t.Fatalf("%d report(s) outlived the machine", left)
|
|
}
|
|
}
|
|
|
|
// "Behind" must mean not running what the mesh would send, not only "failed".
|
|
//
|
|
// novox/hq ADR 0010 names losing "did my change go out?" as the real risk of replacing a pipeline
|
|
// with a comparison. With behind meaning only failed-or-refused, that question was answerable
|
|
// exactly for the machines that broke — and for every machine that worked, the answer was silence
|
|
// whether the change had gone out or not.
|
|
func TestAMachineIsWaitingWhenWhatItWasSentIsNotWhatItShouldBe(t *testing.T) {
|
|
inv := ForTest(t)
|
|
ctx := t.Context()
|
|
anchor, err := inv.AddNode(ctx, "anchor")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if _, err := inv.AddNode(ctx, "laptop"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
// Never sent anything: waiting, and said differently. Nobody has ever asked it to be
|
|
// anything, which is not the same as it being out of date.
|
|
waiting, err := inv.Waiting(ctx, map[string]string{"anchor": "aaa", "laptop": "bbb"})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(waiting) != 2 {
|
|
t.Fatalf("machines that were never sent anything are not waiting: %+v", waiting)
|
|
}
|
|
for _, m := range waiting {
|
|
if !m.Never {
|
|
t.Fatalf("%s was never sent anything and does not say so: %+v", m.Node, m)
|
|
}
|
|
}
|
|
|
|
// Sent what it should be: not waiting.
|
|
if err := inv.RecordSent(ctx, anchor.ID, "aaa"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
waiting, err = inv.Waiting(ctx, map[string]string{"anchor": "aaa", "laptop": "bbb"})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(waiting) != 1 || waiting[0].Node != "laptop" {
|
|
t.Fatalf("a machine sent exactly what it should be is still waiting: %+v", waiting)
|
|
}
|
|
|
|
// The declaration changes: waiting again, and no longer "never".
|
|
waiting, err = inv.Waiting(ctx, map[string]string{"anchor": "ccc", "laptop": "bbb"})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
var found bool
|
|
for _, m := range waiting {
|
|
if m.Node != "anchor" {
|
|
continue
|
|
}
|
|
found = true
|
|
if m.Never {
|
|
t.Fatal("a machine that has been sent something is reported as never told")
|
|
}
|
|
if m.Sent != "aaa" {
|
|
t.Fatalf("what it was last sent was lost: %+v", m)
|
|
}
|
|
}
|
|
if !found {
|
|
t.Fatal("a machine whose declaration changed since it was sent is not waiting")
|
|
}
|
|
}
|
|
|
|
// A machine nobody worked out is not reported as waiting: saying so would invent a comparison.
|
|
func TestAMachineWithNothingComputedForItIsNotWaiting(t *testing.T) {
|
|
inv := ForTest(t)
|
|
ctx := t.Context()
|
|
node, err := inv.AddNode(ctx, "unresolvable")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// It has been sent something before, which is what makes this the case the guard is for: a
|
|
// machine with a digest and nothing computed for it would compare against the empty string
|
|
// and look out of date, when the truth is that nobody worked out what it should be.
|
|
if err := inv.RecordSent(ctx, node.ID, "what-it-got-last-time"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
waiting, err := inv.Waiting(ctx, map[string]string{})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(waiting) != 0 {
|
|
t.Fatalf("a machine the caller could not work out was reported as waiting: %+v", waiting)
|
|
}
|
|
}
|
|
|
|
// **A failure that repeats is told apart from one that just happened.** A node re-applies on a
|
|
// steady interval and reports each time, so a resource that will never apply arrives as the same
|
|
// report over and over — one row, replaced, "failed" at a fresh time — and nothing distinguished
|
|
// it from a failure that goes away by itself (novox/hq 04-ISSUES/065). The row now keeps when the
|
|
// current failure began and how many reports in a row have said it.
|
|
func TestTheSameFailureReportedAgainIsCountedNotRestarted(t *testing.T) {
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "looping")
|
|
same := Doing{Outcome: OutcomeFailed, Failed: []FailedResource{{ID: "img", Error: "no such image"}}}
|
|
|
|
if _, err := inv.RecordDoing(ctx, id, same); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
first, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if first.Times != 1 || first.Since == nil || first.Stuck() {
|
|
t.Fatalf("one failure is one failure, not yet stuck: %+v", first)
|
|
}
|
|
|
|
for range StuckAfter - 1 {
|
|
if _, err := inv.RecordDoing(ctx, id, same); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
}
|
|
again, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if again.Times != StuckAfter || !again.Stuck() {
|
|
t.Fatalf("the same failure %d times is stuck: %+v", StuckAfter, again)
|
|
}
|
|
if !again.Since.Equal(*first.Since) {
|
|
t.Fatalf("the failure began at %s and the row now says %s", *first.Since, *again.Since)
|
|
}
|
|
|
|
// The same resource failing with different words — a duration, a counter — is still the same
|
|
// failure: it is the resource that loops, not the sentence.
|
|
reworded := Doing{Outcome: OutcomeFailed, Failed: []FailedResource{{ID: "img", Error: "no such image (after 31s)"}}}
|
|
if _, err := inv.RecordDoing(ctx, id, reworded); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
still, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if still.Times != StuckAfter+1 || !still.Stuck() {
|
|
t.Fatalf("the same resource failing in other words restarted the count: %+v", still)
|
|
}
|
|
|
|
// A different failure is a new situation, not a longer one.
|
|
other := Doing{Outcome: OutcomeFailed, Failed: []FailedResource{{ID: "svc", Error: "unit not found"}}}
|
|
if _, err := inv.RecordDoing(ctx, id, other); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
changed, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if changed.Times != 1 || changed.Stuck() || !changed.Since.After(*first.Since) && !changed.Since.Equal(*first.Since) {
|
|
t.Fatalf("a new failure starts the count again: %+v", changed)
|
|
}
|
|
|
|
// And a clean apply clears it: the machine is doing what it was told, since nothing.
|
|
if _, err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeApplied, Applied: 2}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
fine, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if fine.Times != 0 || fine.Since != nil || fine.Stuck() {
|
|
t.Fatalf("a machine doing what it was told is not stuck: %+v", fine)
|
|
}
|
|
|
|
// The list of what is wrong carries the count, so `status` can say it.
|
|
if _, err := inv.RecordDoing(ctx, id, same); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 1 || wrong[0].Times != 1 || wrong[0].Since == nil {
|
|
t.Fatalf("got %+v", wrong)
|
|
}
|
|
}
|