From 077ddf0eb885bea74aca0b29cd05b195e06ec780 Mon Sep 17 00:00:00 2001 From: jochen Date: Fri, 9 Oct 2026 13:30:18 +0200 Subject: [PATCH] Judge a send by when a fault began, not when it was last raised (hq issue 348) On 2026-10-09 the control node's resolver refused from 10:57:57 UTC. A node-engine restarted by the 10:59:34 send said its names undecided, that statement cleared the network condition, the next look raised it again after the send, and the gate put back two builds for a fault older than them. - A condition keeps First across a reopening; the gate reads Began. - An undecided network statement (unknown, starting) clears nothing. - D10 counts what a release's tier names as rolling, so a walked node-engine is not core-behind on its own first machine. - D2 raises a resolver that refuses every try at once: a refusal is an answer, not a loaded resolver (issue 277). --- cmd/mesh-controller/gate.go | 7 +- cmd/mesh-controller/issue348_test.go | 212 +++++++++++++++++++++++++ cmd/mesh-controller/machine_network.go | 33 +++- cmd/mesh-controller/probes.go | 51 +++++- internal/conditions/condition.go | 16 ++ internal/conditions/store.go | 15 +- internal/conditions/store_test.go | 53 +++++++ 7 files changed, 370 insertions(+), 17 deletions(-) create mode 100644 cmd/mesh-controller/issue348_test.go diff --git a/cmd/mesh-controller/gate.go b/cmd/mesh-controller/gate.go index f47929cf..7e3a8631 100644 --- a/cmd/mesh-controller/gate.go +++ b/cmd/mesh-controller/gate.go @@ -209,7 +209,8 @@ func judgeHealth(module, component string, m catalogue.Manifest, machine string, return healthNotYet, fmt.Sprintf("%s reported %q", machine, r.Outcome) } // **No new condition about it**: about the machine itself, or naming the module on that machine, - // raised since the judging began. The gate's own are not evidence about the build. + // raised since the judging began. The gate's own are not evidence about the build. A fault reopened + // since, that began before the send, is not new (Began, novox/hq issue 348). if f.judged { if f.openErr != nil { return healthNotYet, "what is wrong cannot be read, so whether the build made anything wrong is not known: " + @@ -219,7 +220,7 @@ func judgeHealth(module, component string, m catalogue.Manifest, machine string, // A wait for a person's new login, or for a directory used as found to be handed over, is the module's // reading, not a fault raised since the send: the gate reads it from the statement below (ADR 0254, // novox/hq issue 339). - if c.Source == gateProbe || c.Raised.Before(since) || c.Kind == kindReloginNeeded || c.Kind == kindUsedAsFound { + if c.Source == gateProbe || c.Began().Before(since) || c.Kind == kindReloginNeeded || c.Kind == kindUsedAsFound { continue } onIt := c.Subject.Machine == machine || slices.Contains(c.Subject.Also, machine) || @@ -331,7 +332,7 @@ func aboutTheMachine(machine string, moved []string, since time.Time, f gateFact aboutIt := c.Subject.Scope == conditions.ScopeMachine && (c.Subject.ID == machine || c.Subject.Machine == machine || slices.Contains(c.Subject.Also, machine)) // A directory used as found waits for a person, whatever the send did (novox/hq issue 339). - if !aboutIt || c.Source == gateProbe || c.Raised.Before(since) || c.Kind == kindUsedAsFound { + if !aboutIt || c.Source == gateProbe || c.Began().Before(since) || c.Kind == kindUsedAsFound { kept = append(kept, c) continue } diff --git a/cmd/mesh-controller/issue348_test.go b/cmd/mesh-controller/issue348_test.go new file mode 100644 index 00000000..39275326 --- /dev/null +++ b/cmd/mesh-controller/issue348_test.go @@ -0,0 +1,212 @@ +package main + +import ( + "errors" + "net" + "os" + "strings" + "syscall" + "testing" + "time" + + "github.com/novox/mesh-controller/internal/catalogue" + "github.com/novox/mesh-controller/internal/conditions" + "github.com/novox/mesh-controller/internal/inventory" + "github.com/novox/mesh-controller/internal/lease" + "github.com/novox/mesh-controller/internal/link" +) + +// novox/hq issue 348: on 2026-10-09 the control node's resolver stopped answering on its private address +// at 10:57:57 UTC; the machine's network condition was raised at 10:58:45. The node-engine and the +// controller were sent at 10:59:34. The new node-engine's first statement judged the names once — unknown, +// "one look failed; a second decides" — and that statement cleared the condition, though its last evidence +// still said "connection refused". The next look raised it again at 11:00:23, after the send, and both +// builds failed their gate at 11:10 with "raised since it was sent" and were put back, for a fault that +// began before they were sent. + +// namesRefused is the control node's names part as its node-engine said it in the outage: its own +// resolver, at its own address, refusing. +func namesRefused(state string, streak int) link.NetworkPart { + p := link.NetworkPart{Part: link.PartNames, State: state, Since: h0, Streak: streak} + if state == link.StateUnhealthy { + p.Reason = "1 of its 2 resolvers do not answer as the mesh's do" + p.Said = "10.77.0.1 — anchor.internal (IPv4): read udp 10.77.0.1:35244->10.77.0.1:53: read: connection refused" + p.Toward = []string{"10.77.0.1"} + } + return p +} + +func networkSaying(parts ...link.NetworkPart) *link.NetworkHealth { + state := link.StateHealthy + for _, p := range parts { + switch { + case p.State == link.StateUnhealthy: + state = link.StateUnhealthy + case p.State == link.StateUnknown && state == link.StateHealthy: + state = link.StateUnknown + } + } + return &link.NetworkHealth{State: state, Since: h0, Parts: parts} +} + +// TestReplay348 replays the statements of the outage: the condition raised before the send is not +// cleared by the restarted engine's first, undecided statement, and the gate does not count it against +// the builds sent after it began. +func TestReplay348(t *testing.T) { + open := aMesh(t) + ctx := t.Context() + inv, k := open.inventory, conditionsFrom + say := func(at time.Time, n *link.NetworkHealth) { + t.Helper() + if err := stateHealth(ctx, inv, k, "anchor", link.Health{Contract: link.ReadinessContract, At: at, Network: n}, at); err != nil { + t.Fatal(err) + } + } + network := func() (conditions.Condition, bool) { + t.Helper() + list, err := k.Open(ctx) + if err != nil { + t.Fatal(err) + } + for _, c := range list { + if c.Key == "machine.anchor.network" { + return c, true + } + } + return conditions.Condition{}, false + } + + // 10:58:45 — the second failing look: raised. + say(h0, networkSaying(namesRefused(link.StateUnhealthy, 2))) + raised, ok := network() + if !ok { + t.Fatal("the resolver refusing on the control node raised nothing") + } + time.Sleep(5 * time.Millisecond) + sent := time.Now().UTC() + time.Sleep(5 * time.Millisecond) + + // 10:59:42 — the restarted engine's first statement: one look failed, a second decides. + say(h0.Add(time.Minute), networkSaying(namesRefused(link.StateUnknown, 1))) + if _, ok := network(); !ok { + t.Fatal("a statement that judged nothing yet cleared the condition: the restarted engine's first look " + + "said the fault was gone while it still refused") + } + // 11:00:23 — its second look: unhealthy again, the same raising. + say(h0.Add(2*time.Minute), networkSaying(namesRefused(link.StateUnhealthy, 2))) + again, ok := network() + if !ok || !again.Began().Equal(raised.Raised) { + t.Fatalf("the same fault was said as one that began at %s (raised first at %s)", again.Began(), raised.Raised) + } + + // The gate on the control node, for a build sent after the fault began. + open2, err := k.Open(ctx) + if err != nil { + t.Fatal(err) + } + f := gateFacts{judged: true, open: open2} + if w := aboutTheMachine("anchor", []string{"mesh-host"}, sent, f); w.whole != "" || len(w.on) != 0 { + t.Fatalf("a fault from before the send held the build: %+v", w) + } + + // Decided healthy: cleared. + say(h0.Add(3*time.Minute), networkSaying(link.NetworkPart{Part: link.PartNames, State: link.StateHealthy, Since: h0})) + if c, ok := network(); ok { + t.Fatalf("a statement that decides the names healthy left %s open", c.Key) + } +} + +// A reopening is the same fault: it began when it was first raised, and the gate reads when it began. The +// controller that cleared it on an undecided statement — or any clearing and raising within ReopenWithin — +// no longer makes a fault from before a send into one raised since it. +func TestAReopenedFaultFromBeforeTheSendIsNotRaisedSinceIt(t *testing.T) { + since := h0 + c := conditions.Condition{Key: "machine.anchor.network", Kind: kindMachineNetwork, + Subject: conditions.Subject{Scope: conditions.ScopeMachine, ID: "anchor", Machine: "anchor"}, + Summary: "anchor's network is not healthy", Source: sourceNetwork, Count: 2, + First: since.Add(-time.Minute), Raised: since.Add(time.Minute)} + f := gateFacts{judged: true, open: []conditions.Condition{c}} + if w := aboutTheMachine("anchor", []string{"mesh-controller"}, since, f); w.whole != "" || len(w.on) != 0 { + t.Fatalf("a fault reopened after the send, first raised before it, held the machine: %+v", w) + } + // The same raised first after the send is the send's to answer for. + c.First = since.Add(30 * time.Second) + f.open = []conditions.Condition{c} + if w := aboutTheMachine("anchor", []string{"mesh-controller"}, since, f); w.whole == "" { + t.Fatalf("a fault that began after the send held nothing: %+v", w) + } + // And a module's own: judgeHealth reads it the same way. + at := since.Add(2 * time.Minute) + g := gateFacts{judged: true, now: at, reports: map[string]inventory.Reported{"anchor": {Node: "anchor", + Outcome: inventory.OutcomeApplied, At: &at, Current: true}}, engines: map[string]string{}, + served: map[string]served{}, rolledBack: map[string][]lease.Rollback{}, + open: []conditions.Condition{{Key: "provider.app.anchor.x.failing", Subject: conditions.Subject{ + Scope: conditions.ScopeProvider, ID: "app.anchor.x", Machine: "anchor"}, Summary: "failing", + First: since.Add(-time.Hour), Raised: since.Add(time.Minute)}}} + if h, why := judgeHealth("app", "", catalogue.Manifest{Module: "app"}, "anchor", since, g); strings.HasPrefix(why, "raised since it was sent") { + t.Fatalf("a module's own fault, reopened after its send and first raised before it: %v (%s)", h, why) + } + g.open[0].First = time.Time{} + if _, why := judgeHealth("app", "", catalogue.Manifest{Module: "app"}, "anchor", since, g); !strings.HasPrefix(why, "raised since it was sent") { + t.Fatalf("a module's own fault raised after its send was not counted: %s", why) + } +} + +// An undecided statement — unknown or starting — keeps the machine's network conditions; a decided one +// does not. Pure. +func TestAnUndecidedStatementKeepsTheMachinesNetworkConditions(t *testing.T) { + f := netFacts(map[string]*inventory.NetworkHealth{ + "anchor": aNetwork(link.StateUnknown, inventory.NetworkPart{Part: link.PartNames, State: link.StateUnknown, Streak: 1}), + "laptop": aNetwork(link.StateHealthy, inventory.NetworkPart{Part: link.PartNames, State: link.StateHealthy}), + "spare": aNetwork(link.StateUnhealthy, noRoute), + }) + got := undecidedMachines(f) + if !got["anchor"] || got["laptop"] || got["spare"] || len(got) != 1 { + t.Fatalf("undecided: %v", got) + } +} + +// A release walks its modules without a record per module: D10 counts what its tier names as rolling, +// so the node-engine a release walks is not "behind, and no plan is rolling it out" on its first machine. +func TestAReleaseRollsOutWhatItsTierNames(t *testing.T) { + plans := []inventory.Plan{ + {ID: "release-1", State: inventory.PlanRolling, Tiers: [][]string{{"mesh-host"}}, Modules: map[string]*inventory.PlanModule{}}, + {ID: "plan-2", State: inventory.PlanRolling, Modules: map[string]*inventory.PlanModule{"letta": {}}}, + } + got := rollingModules(plans) + if !got["mesh-host"] || !got["letta"] || len(got) != 2 { + t.Fatalf("rolling: %v", got) + } +} + +// A resolver that refuses every try — nothing listens on its port — is said at once, not held for the +// next run five minutes later: a refusal is the machine's answer, not a loaded resolver (issue 277). +func TestAResolverThatRefusesEveryTryIsSaidAtOnce(t *testing.T) { + conn, err := net.ListenPacket("udp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + _, port, _ := net.SplitHostPort(conn.LocalAddr().String()) + _ = conn.Close() // nothing listens there now: every question is refused + quickResolvers(t, port) + got := askEveryResolver(t.Context(), map[string]string{"anchor.internal": "127.0.0.1"}, anchorOnTheNetwork, "internal") + if len(got) != 1 { + t.Fatalf("%+v", got) + } + if got[0].Confirm || !strings.Contains(got[0].Said, "refused") { + t.Fatalf("a resolver refusing every try was held for the next run: %+v", got[0]) + } +} + +func TestOnlyARefusalOnEveryTryIsAllRefused(t *testing.T) { + refused := &net.OpError{Op: "read", Net: "udp", Err: os.NewSyscallError("read", syscall.ECONNREFUSED)} + if !(resolverAsked{errs: []error{refused, refused}}).allRefused() { + t.Fatal("refused on every try is not all refused") + } + if (resolverAsked{errs: []error{refused, errors.New("i/o timeout")}}).allRefused() { + t.Fatal("a timeout among the tries is all refused") + } + if (resolverAsked{answered: true, errs: []error{refused}}).allRefused() || (resolverAsked{}).allRefused() { + t.Fatal("an answered question, or one never asked, is all refused") + } +} diff --git a/cmd/mesh-controller/machine_network.go b/cmd/mesh-controller/machine_network.go index 8fe0ce18..fcb081cf 100644 --- a/cmd/mesh-controller/machine_network.go +++ b/cmd/mesh-controller/machine_network.go @@ -98,8 +98,9 @@ func judgeNetworks(ctx context.Context, inv *inventory.Inventory, k *conditions. problems = append(problems, err.Error()) } } + undecided := undecidedMachines(f) for _, c := range open { - if !slices.Contains(networkKinds, c.Kind) || said[c.Key] { + if !slices.Contains(networkKinds, c.Kind) || said[c.Key] || undecided[c.Subject.Machine] { continue } why := "no machine says it any more" @@ -116,6 +117,36 @@ func judgeNetworks(ctx context.Context, inv *inventory.Inventory, k *conditions. return nil } +// undecidedMachines is every machine whose newest statement has a part not yet judged: starting, or +// unknown — one look failed and a second decides (novox/hq issue 348). Such a statement does not say a +// fault is gone, so an open network condition about that machine is kept until a statement decides. +// +// On 2026-10-09 the control node's resolver refused every question from 10:58 to 11:18 UTC. A build of +// the node-engine sent at 10:59:34 restarted it; its first statement after the restart judged the names +// once (unknown, "one look failed; a second decides"), and that statement cleared the control node's +// network condition while its last evidence still said "connection refused". The second look raised it +// again forty seconds later — after the send — and the gate failed the build for a fault from before it. +// Pure. +func undecidedMachines(f networkFacts) map[string]bool { + out := map[string]bool{} + for m, h := range f.healths { + if h.Network == nil { + continue + } + if h.Network.State == link.StateUnknown || h.Network.State == link.StateStarting { + out[m] = true + continue + } + for _, p := range h.Network.Parts { + if p.State == link.StateUnknown || p.State == link.StateStarting { + out[m] = true + break + } + } + } + return out +} + // pointed is one machine's failing part that points at another machine. type pointed struct { from string diff --git a/cmd/mesh-controller/probes.go b/cmd/mesh-controller/probes.go index 6e034d07..f236a0ac 100644 --- a/cmd/mesh-controller/probes.go +++ b/cmd/mesh-controller/probes.go @@ -11,6 +11,7 @@ import ( "sort" "strings" "sync" + "syscall" "time" "github.com/nats-io/nats.go" @@ -216,6 +217,8 @@ func askEveryResolver(ctx context.Context, resolvers map[string]string, places [ type verdict struct { wrong, unanswered, said []string asked int + // refused says every try of every unanswered question was refused (nothing listens there). + refused bool } byHolder := map[string]*verdict{} for _, q := range questions { @@ -232,6 +235,7 @@ func askEveryResolver(ctx context.Context, resolvers map[string]string, places [ v.wrong = append(v.wrong, words) v.said = append(v.said, said) default: + v.refused = (len(v.unanswered) == 0 || v.refused) && q.asked.allRefused() v.unanswered = append(v.unanswered, words) v.said = append(v.said, said) } @@ -251,8 +255,11 @@ func askEveryResolver(ctx context.Context, resolvers map[string]string, places [ o.Summary = fmt.Sprintf("the mesh's resolver on %s does not answer machine names as it must: %s%s", node, v.wrong[0], andMore(len(v.wrong)-1)) } else { - // Nothing but silence: held for the next run, which raises it if the resolver is still silent. - o.Confirm = true + // Nothing but silence: held for the next run, which raises it if the resolver is still silent — + // unless every try of every question was refused. A refusal is an answer from the machine: + // nothing listens there. It is not a loaded resolver slow to answer (issue 277), and holding it + // for the next run cost the control node's resolver five minutes on 2026-10-09 (issue 348). + o.Confirm = !v.refused o.Summary = fmt.Sprintf("the mesh's resolver on %s does not answer: %d of the %d question(s) about the "+ "machines' names went unanswered, each asked %d times — the first, %s", node, len(v.unanswered), v.asked, resolverTries, v.unanswered[0]) @@ -341,6 +348,19 @@ type resolverAsked struct { errs []error } +// allRefused says every try was refused: the resolver's machine said nothing listens on its port. +func (a resolverAsked) allRefused() bool { + if a.answered || len(a.errs) == 0 { + return false + } + for _, err := range a.errs { + if !errors.Is(err, syscall.ECONNREFUSED) { + return false + } + } + return true +} + // askResolverPatiently asks one question until it is answered as right says, at most resolverTries // times. A wrong answer is asked again too — a resolver restarting may say NXDOMAIN for a moment — and // stands if no try answers rightly. @@ -1040,12 +1060,7 @@ func probeCoreBuilds(ctx context.Context, d *doctor) ([]conditions.Observation, return nil, err } hostVersions := deliveredVersions(shelf[hostModule]) - rolling := map[string]bool{} - for _, p := range plans { - for m := range p.Modules { - rolling[m] = true - } - } + rolling := rollingModules(plans) nodes, err := inv.Nodes(ctx) if err != nil { return nil, err @@ -1103,6 +1118,26 @@ func probeCoreBuilds(ctx context.Context, d *doctor) ([]conditions.Observation, return out, nil } +// rollingModules is every module an open plan is rolling out: those it keeps a record of, and those its +// tiers name. **A release keeps no record per module** — its walk is per machine, its modules only in its +// tier — so reading the records alone, D10 said "no plan is rolling them out" about the node-engine while +// a release walked it, and the gate on that release's first machine waited on what its own send causes +// (novox/hq issue 348). Pure. +func rollingModules(plans []inventory.Plan) map[string]bool { + rolling := map[string]bool{} + for _, p := range plans { + for m := range p.Modules { + rolling[m] = true + } + for _, tier := range p.Tiers { + for _, m := range tier { + rolling[m] = true + } + } + } + return rolling +} + // deliveredVersions are the versions a module's registered build is delivered as: the last element // of every resource path under a `versions/` directory, which registration filled from the artifact's // digest (catalogue `${version}`). The node-engine names itself by that directory. diff --git a/internal/conditions/condition.go b/internal/conditions/condition.go index cc253a6a..ce408349 100644 --- a/internal/conditions/condition.go +++ b/internal/conditions/condition.go @@ -145,6 +145,11 @@ type Condition struct { // Raised is when it was first observed this time; LastObserved the newest observation. Raised time.Time `json:"raised"` LastObserved time.Time `json:"last-observed"` + // First is when this fault was first raised, a reopening within ReopenWithin counted as the same + // fault: Raised of the first raising. Absent on one written before it was kept, and on a first + // raising, where it is Raised (Began). A reopening is the same fault, so what asks when a fault began + // — the gate, asking whether it began after a send — reads Began, never Raised (novox/hq issue 348). + First time.Time `json:"first,omitzero"` // Observations is how many times it was observed since raised. Observations int `json:"observations"` // Count is how many times it has been raised, a reopening within ReopenWithin counted. @@ -274,3 +279,14 @@ func Order(list []Condition) { return list[i].Key < list[j].Key }) } + +// Began is when this fault began as the mesh knows it: its first raising, a reopening within +// ReopenWithin being the same fault again (novox/hq issue 348). A condition that cleared on a statement +// that judged nothing yet — a node-engine just restarted — and was raised again a minute later began +// when it was first raised, not at the reopening. +func (c Condition) Began() time.Time { + if !c.First.IsZero() && c.First.Before(c.Raised) { + return c.First + } + return c.Raised +} diff --git a/internal/conditions/store.go b/internal/conditions/store.go index cf016f4a..f4935d7d 100644 --- a/internal/conditions/store.go +++ b/internal/conditions/store.go @@ -77,8 +77,10 @@ type Keeper struct { } type clearing struct { - at time.Time - count int + at time.Time + count int + // began is when the fault that cleared began, so its reopening keeps it (Condition.Began). + began time.Time silenced *Silence // tried is what healers tried before it cleared: a reopening is the same fault, and what was // tried on it is still what was tried. @@ -120,8 +122,8 @@ func NewKeeper(ctx context.Context, o Options) *Keeper { if recent, err := k.history.Since(ctx, k.now().Add(-ReopenWithin)); err == nil { for _, e := range recent { if e.Change == ChangeCleared { - k.cleared[e.Key] = clearing{at: e.At, count: e.Condition.Count, silenced: e.Condition.Silenced, - tried: e.Condition.Tried} + k.cleared[e.Key] = clearing{at: e.At, count: e.Condition.Count, began: e.Condition.Began(), + silenced: e.Condition.Silenced, tried: e.Condition.Tried} } } } else { @@ -197,6 +199,9 @@ func (k *Keeper) Observe(ctx context.Context, o Observation) (Condition, error) k.mu.Lock() if before, ok := k.cleared[key]; ok && now.Sub(before.at) <= ReopenWithin { c.Count, change = before.count+1, ChangeReopened + if !before.began.IsZero() && before.began.Before(now) { + c.First = before.began + } // A silence a person gave the condition before it cleared still holds: they said // they knew, and the same fault again ten minutes later is what they knew about. if before.silenced != nil && now.Before(before.silenced.Until) { @@ -313,7 +318,7 @@ func (k *Keeper) ClearSaying(ctx context.Context, key, why, resolved string) (bo } now := k.now().UTC() k.mu.Lock() - k.cleared[key] = clearing{at: now, count: c.Count, silenced: c.Silenced, tried: c.Tried} + k.cleared[key] = clearing{at: now, count: c.Count, began: c.Began(), silenced: c.Silenced, tried: c.Tried} k.mu.Unlock() k.tell(Event{Condition: c, At: now, Change: ChangeCleared, Why: why, Cleared: &now}) return true, nil diff --git a/internal/conditions/store_test.go b/internal/conditions/store_test.go index 91d1fbba..ab2fe986 100644 --- a/internal/conditions/store_test.go +++ b/internal/conditions/store_test.go @@ -378,3 +378,56 @@ func TestATransitionIsOfferedAgainWhileTheBusIsAway(t *testing.T) { t.Fatalf("said %+v, unsaid %d", said, k.Unsaid()) } } + +// **A reopening keeps when the fault began** (novox/hq issue 348): Raised is the reopening, Began the +// first raising — also from a keeper that read what cleared from the history — and past the window a +// raising is a new fault that begins then. +func TestAReopeningKeepsWhenTheFaultBegan(t *testing.T) { + k, store, told, c := keeper(t) + ctx := t.Context() + first, err := k.Observe(ctx, silent("ace")) + if err != nil { + t.Fatal(err) + } + if !first.Began().Equal(first.Raised) || !first.First.IsZero() { + t.Fatalf("a first raising began at %s, raised %s, first %s", first.Began(), first.Raised, first.First) + } + if _, err := k.Clear(ctx, "machine.ace.silent", "heard again"); err != nil { + t.Fatal(err) + } + c.pass(time.Minute) + again, err := k.Observe(ctx, silent("ace")) + if err != nil { + t.Fatal(err) + } + if !again.Raised.After(first.Raised) || !again.Began().Equal(first.Raised) { + t.Fatalf("reopened: raised %s, began %s; first raised %s", again.Raised, again.Began(), first.Raised) + } + // Through a restarted controller, reading the clearing from the history. + if _, err := k.Clear(ctx, "machine.ace.silent", "heard again"); err != nil { + t.Fatal(err) + } + settled(t, told, 4) + k.Close(context.Background()) + next := NewKeeper(ctx, Options{Store: store, History: store, Now: c.now}) + defer next.Close(context.Background()) + c.pass(time.Minute) + third, err := next.Observe(ctx, silent("ace")) + if err != nil { + t.Fatal(err) + } + if !third.Began().Equal(first.Raised) { + t.Fatalf("after a restart the reopening began at %s, not %s", third.Began(), first.Raised) + } + if _, err := next.Clear(ctx, "machine.ace.silent", "heard again"); err != nil { + t.Fatal(err) + } + c.pass(ReopenWithin + time.Minute) + fourth, err := next.Observe(ctx, silent("ace")) + if err != nil { + t.Fatal(err) + } + if !fourth.Began().Equal(fourth.Raised) { + t.Fatalf("past the window the fault began at %s, raised %s", fourth.Began(), fourth.Raised) + } +}