The mesh kept one report per machine, replaced, so a resource nothing can ever apply looked like a failure that had just happened, every reconcile interval, for ever. The row now keeps when the current failure began and how many reports in a row have said it — the same outcome, refusal and failed resources; anything different starts again and a clean apply clears it. Three make the machine stuck, and status says so beside the failure, in words and in JSON (novox/hq 04-ISSUES/065, ADR 0090).
324 lines
10 KiB
Go
324 lines
10 KiB
Go
package inventory
|
|
|
|
import (
|
|
"context"
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// What each machine did with what it was last sent.
|
|
//
|
|
// Until this, a refusal or a failure moved last_seen and the reason went to a log line — so
|
|
// "which machine is not doing what it was told" had no answer the next morning, which is the
|
|
// question a mesh exists to answer.
|
|
|
|
func nodeNamed(t *testing.T, inv *Inventory, name string) string {
|
|
t.Helper()
|
|
n, err := inv.AddNode(context.Background(), name)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
return n.ID
|
|
}
|
|
|
|
func TestRefusedAndFailedAreDifferentSituations(t *testing.T) {
|
|
// Refused means the machine is exactly as it was, and what is wrong is in what was sent.
|
|
// Failed means it is in a state nobody declared, and what is wrong is on the machine. One
|
|
// word for both would make the report say less than the node did.
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
refuser := nodeNamed(t, inv, "refuser")
|
|
failer := nodeNamed(t, inv, "failer")
|
|
|
|
if err := inv.RecordDoing(ctx, refuser, Doing{
|
|
Outcome: OutcomeRefused, Refused: "resource \"x\": a file needs a path",
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := inv.RecordDoing(ctx, failer, Doing{
|
|
Outcome: OutcomeFailed,
|
|
Failed: []FailedResource{{ID: "svc", Error: "unit not found"}},
|
|
Applied: 4,
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 2 {
|
|
t.Fatalf("got %d", len(wrong))
|
|
}
|
|
by := map[string]Doing{}
|
|
for _, d := range wrong {
|
|
by[d.Node] = d
|
|
}
|
|
if by["refuser"].Outcome != OutcomeRefused || !strings.Contains(by["refuser"].Refused, "needs a path") {
|
|
t.Fatalf("got %+v", by["refuser"])
|
|
}
|
|
if by["failer"].Outcome != OutcomeFailed || len(by["failer"].Failed) != 1 {
|
|
t.Fatalf("got %+v", by["failer"])
|
|
}
|
|
// And how much DID work, because "four of five" and "none of five" are different machines.
|
|
if by["failer"].Applied != 4 {
|
|
t.Fatalf("what did apply was not kept: %+v", by["failer"])
|
|
}
|
|
}
|
|
|
|
func TestAMachineDoingWhatItWasToldIsNotOnTheList(t *testing.T) {
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "fine")
|
|
if err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeApplied, Applied: 6}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 0 {
|
|
t.Fatalf("a machine that did as it was told is listed as wrong: %+v", wrong)
|
|
}
|
|
// It is still answerable about, which is a different question.
|
|
d, said, err := inv.DoingOf(ctx, "fine")
|
|
if err != nil || !said {
|
|
t.Fatalf("said=%v err=%v", said, err)
|
|
}
|
|
if d.Wrong() || d.Applied != 6 {
|
|
t.Fatalf("got %+v", d)
|
|
}
|
|
}
|
|
|
|
func TestTheLastReportReplacesTheOneBefore(t *testing.T) {
|
|
// The question is the machine's CURRENT state. "This failed an hour ago and then succeeded"
|
|
// is not a machine anybody needs to look at, and a table of every report would bury the ones
|
|
// that matter under the ones that do not.
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "recovered")
|
|
if err := inv.RecordDoing(ctx, id, Doing{
|
|
Outcome: OutcomeFailed, Failed: []FailedResource{{ID: "a", Error: "no"}},
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeApplied, Applied: 3}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 0 {
|
|
t.Fatalf("a machine that recovered is still listed as wrong: %+v", wrong)
|
|
}
|
|
d, _, _ := inv.DoingOf(ctx, "recovered")
|
|
if len(d.Failed) != 0 {
|
|
t.Fatalf("the old failure survived: %+v", d)
|
|
}
|
|
}
|
|
|
|
func TestAMachineThatHasSaidNothingIsNotWrong(t *testing.T) {
|
|
// It may be new, switched off, or unreachable. None of those is a machine that tried and
|
|
// could not, and treating silence as failure would have somebody debug a machine that has
|
|
// simply never been sent anything.
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
nodeNamed(t, inv, "silent")
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 0 {
|
|
t.Fatalf("silence was read as failure: %+v", wrong)
|
|
}
|
|
if _, said, err := inv.DoingOf(ctx, "silent"); err != nil || said {
|
|
t.Fatalf("a machine that never reported was reported about: said=%v err=%v", said, err)
|
|
}
|
|
}
|
|
|
|
func TestWhatANodeSaidGoesWhenTheNodeDoes(t *testing.T) {
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "leaving")
|
|
if err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeFailed}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if _, err := inv.store.Pool().Exec(ctx, `delete from node where name = 'leaving'`); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
var left int
|
|
if err := inv.store.Pool().QueryRow(ctx, `select count(*) from node_report`).Scan(&left); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if left != 0 {
|
|
t.Fatalf("%d report(s) outlived the machine", left)
|
|
}
|
|
}
|
|
|
|
// "Behind" must mean not running what the mesh would send, not only "failed".
|
|
//
|
|
// novox/hq ADR 0010 names losing "did my change go out?" as the real risk of replacing a pipeline
|
|
// with a comparison. With behind meaning only failed-or-refused, that question was answerable
|
|
// exactly for the machines that broke — and for every machine that worked, the answer was silence
|
|
// whether the change had gone out or not.
|
|
func TestAMachineIsWaitingWhenWhatItWasSentIsNotWhatItShouldBe(t *testing.T) {
|
|
inv := ForTest(t)
|
|
ctx := t.Context()
|
|
anchor, err := inv.AddNode(ctx, "anchor")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if _, err := inv.AddNode(ctx, "laptop"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
// Never sent anything: waiting, and said differently. Nobody has ever asked it to be
|
|
// anything, which is not the same as it being out of date.
|
|
waiting, err := inv.Waiting(ctx, map[string]string{"anchor": "aaa", "laptop": "bbb"})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(waiting) != 2 {
|
|
t.Fatalf("machines that were never sent anything are not waiting: %+v", waiting)
|
|
}
|
|
for _, m := range waiting {
|
|
if !m.Never {
|
|
t.Fatalf("%s was never sent anything and does not say so: %+v", m.Node, m)
|
|
}
|
|
}
|
|
|
|
// Sent what it should be: not waiting.
|
|
if err := inv.RecordSent(ctx, anchor.ID, "aaa"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
waiting, err = inv.Waiting(ctx, map[string]string{"anchor": "aaa", "laptop": "bbb"})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(waiting) != 1 || waiting[0].Node != "laptop" {
|
|
t.Fatalf("a machine sent exactly what it should be is still waiting: %+v", waiting)
|
|
}
|
|
|
|
// The declaration changes: waiting again, and no longer "never".
|
|
waiting, err = inv.Waiting(ctx, map[string]string{"anchor": "ccc", "laptop": "bbb"})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
var found bool
|
|
for _, m := range waiting {
|
|
if m.Node != "anchor" {
|
|
continue
|
|
}
|
|
found = true
|
|
if m.Never {
|
|
t.Fatal("a machine that has been sent something is reported as never told")
|
|
}
|
|
if m.Sent != "aaa" {
|
|
t.Fatalf("what it was last sent was lost: %+v", m)
|
|
}
|
|
}
|
|
if !found {
|
|
t.Fatal("a machine whose declaration changed since it was sent is not waiting")
|
|
}
|
|
}
|
|
|
|
// A machine nobody worked out is not reported as waiting: saying so would invent a comparison.
|
|
func TestAMachineWithNothingComputedForItIsNotWaiting(t *testing.T) {
|
|
inv := ForTest(t)
|
|
ctx := t.Context()
|
|
node, err := inv.AddNode(ctx, "unresolvable")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// It has been sent something before, which is what makes this the case the guard is for: a
|
|
// machine with a digest and nothing computed for it would compare against the empty string
|
|
// and look out of date, when the truth is that nobody worked out what it should be.
|
|
if err := inv.RecordSent(ctx, node.ID, "what-it-got-last-time"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
waiting, err := inv.Waiting(ctx, map[string]string{})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(waiting) != 0 {
|
|
t.Fatalf("a machine the caller could not work out was reported as waiting: %+v", waiting)
|
|
}
|
|
}
|
|
|
|
// **A failure that repeats is told apart from one that just happened.** A node re-applies on a
|
|
// steady interval and reports each time, so a resource that will never apply arrives as the same
|
|
// report over and over — one row, replaced, "failed" at a fresh time — and nothing distinguished
|
|
// it from a failure that goes away by itself (novox/hq 04-ISSUES/065). The row now keeps when the
|
|
// current failure began and how many reports in a row have said it.
|
|
func TestTheSameFailureReportedAgainIsCountedNotRestarted(t *testing.T) {
|
|
inv := fresh(t)
|
|
ctx := context.Background()
|
|
id := nodeNamed(t, inv, "looping")
|
|
same := Doing{Outcome: OutcomeFailed, Failed: []FailedResource{{ID: "img", Error: "no such image"}}}
|
|
|
|
if err := inv.RecordDoing(ctx, id, same); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
first, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if first.Times != 1 || first.Since == nil || first.Stuck() {
|
|
t.Fatalf("one failure is one failure, not yet stuck: %+v", first)
|
|
}
|
|
|
|
for range StuckAfter - 1 {
|
|
if err := inv.RecordDoing(ctx, id, same); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
}
|
|
again, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if again.Times != StuckAfter || !again.Stuck() {
|
|
t.Fatalf("the same failure %d times is stuck: %+v", StuckAfter, again)
|
|
}
|
|
if !again.Since.Equal(*first.Since) {
|
|
t.Fatalf("the failure began at %s and the row now says %s", *first.Since, *again.Since)
|
|
}
|
|
|
|
// A different failure is a new situation, not a longer one.
|
|
other := Doing{Outcome: OutcomeFailed, Failed: []FailedResource{{ID: "svc", Error: "unit not found"}}}
|
|
if err := inv.RecordDoing(ctx, id, other); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
changed, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if changed.Times != 1 || changed.Stuck() || !changed.Since.After(*first.Since) && !changed.Since.Equal(*first.Since) {
|
|
t.Fatalf("a new failure starts the count again: %+v", changed)
|
|
}
|
|
|
|
// And a clean apply clears it: the machine is doing what it was told, since nothing.
|
|
if err := inv.RecordDoing(ctx, id, Doing{Outcome: OutcomeApplied, Applied: 2}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
fine, _, err := inv.DoingOf(ctx, "looping")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if fine.Times != 0 || fine.Since != nil || fine.Stuck() {
|
|
t.Fatalf("a machine doing what it was told is not stuck: %+v", fine)
|
|
}
|
|
|
|
// The list of what is wrong carries the count, so `status` can say it.
|
|
if err := inv.RecordDoing(ctx, id, same); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
wrong, err := inv.NotDoingWhatTheyWereTold(ctx)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(wrong) != 1 || wrong[0].Times != 1 || wrong[0].Since == nil {
|
|
t.Fatalf("got %+v", wrong)
|
|
}
|
|
}
|