With twelve consumers of the database provision, one provider down would be twelve conditions for one fault and twelve gates failed for something none of them did. A consumer's check names the provision it exercises; while the provider composed for it — its recorded binding, or the machine its credential comes from — is unhealthy on the record, what that check finds raises nothing of its own: the provider's condition lists it as waiting and is urgent, and the consumer's gate waits, past its bound too, rather than putting a build back. Liveness findings and checks naming no provision stay the consumer's own, and once the provider is healthy a consumer still failing is raised at once.
326 lines
12 KiB
Go
326 lines
12 KiB
Go
package main
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"sort"
|
|
"strings"
|
|
"sync/atomic"
|
|
"time"
|
|
|
|
"github.com/novox/mesh-controller/internal/catalogue"
|
|
"github.com/novox/mesh-controller/internal/conditions"
|
|
"github.com/novox/mesh-controller/internal/inventory"
|
|
"github.com/novox/mesh-controller/internal/link"
|
|
)
|
|
|
|
// A module says how it is healthy, and the node-engine judges it (novox/hq ADR 0240, to-be 48 §4 and §5,
|
|
// Phase A).
|
|
//
|
|
// **The node-engine owns every verdict; the controller keeps the last word and raises the condition.** A
|
|
// machine states, in every report and as an event between reports, the state of every long-running
|
|
// resource it runs for a module. The controller keeps the newest statement per machine (node_health), and
|
|
// raises `module.<module>.<machine>.unhealthy` when two statements in a row say a resource of the module
|
|
// is unhealthy — one statement is listed as unconfirmed, as the self-check does a finding one look can be
|
|
// wrong about (to-be 45 §4, issue 277) — and clears it on the first that does not. The release gate reads
|
|
// the stated health: a judging passes a module only when every long-running resource of it on that
|
|
// machine is stated healthy, so a resource still starting is not yet a pass.
|
|
//
|
|
// **An engine older than the judging states nothing**, and its machine's health is not known: never
|
|
// healthy, never a reason to raise anything, and the gate judges it as it did before.
|
|
|
|
// The condition a module's health raises.
|
|
const (
|
|
kindModuleUnhealthy = "module-unhealthy"
|
|
// sourceHealth is what raised it: the machine's own statement.
|
|
sourceHealth = "health"
|
|
// moduleUnhealthyUrgentAfter is how long it stands before it is urgent (to-be 48 §4).
|
|
moduleUnhealthyUrgentAfter = 4 * time.Hour
|
|
// moduleUnhealthyAfter is how many statements in a row raise it.
|
|
moduleUnhealthyAfter = 2
|
|
)
|
|
|
|
// healthRefused counts the statements refused as older than the one kept, for the log and a test.
|
|
var healthRefused atomic.Int64
|
|
|
|
// moduleHealth keeps what the machines state, for the link (link.Healths).
|
|
type moduleHealth struct {
|
|
inv *inventory.Inventory
|
|
keeper func() *conditions.Keeper
|
|
}
|
|
|
|
func (m moduleHealth) Stated(ctx context.Context, node string, h link.Health) error {
|
|
return stateHealth(ctx, m.inv, m.keeper(), node, h, time.Now())
|
|
}
|
|
|
|
// stateHealth keeps one machine's statement and raises or clears its modules' conditions from it. An
|
|
// older statement than the one kept is refused, by when the engine looked.
|
|
func stateHealth(ctx context.Context, inv *inventory.Inventory, k *conditions.Keeper, node string, h link.Health,
|
|
now time.Time) error {
|
|
if h.Contract == 0 {
|
|
return nil
|
|
}
|
|
prev, had, err := inv.HealthOf(ctx, node)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if had && h.At.Before(prev.SaidAt) {
|
|
healthRefused.Add(1)
|
|
return nil
|
|
}
|
|
unhealthy := map[string][]inventory.ResourceHealth{}
|
|
resources := make([]inventory.ResourceHealth, 0, len(h.Resources))
|
|
for _, r := range h.Resources {
|
|
kept := inventory.ResourceHealth{Module: r.Module, Resource: r.Resource, Kind: r.Kind, Target: r.Target,
|
|
State: r.State, Reason: r.Reason, Since: r.Since, Streak: r.Streak, Restarts: r.Restarts,
|
|
Check: r.Check, Needs: r.Needs}
|
|
resources = append(resources, kept)
|
|
if r.State == link.StateUnhealthy && r.Module != "" {
|
|
unhealthy[r.Module] = append(unhealthy[r.Module], kept)
|
|
}
|
|
}
|
|
streaks := map[string]int{}
|
|
for module := range unhealthy {
|
|
streaks[module] = prev.Streaks[module] + 1
|
|
}
|
|
stored, err := inv.RecordHealth(ctx, inventory.NodeHealth{Node: node, Contract: h.Contract, SaidAt: h.At,
|
|
HeardAt: now, Resources: resources, Streaks: streaks})
|
|
if err != nil || !stored {
|
|
if err == nil {
|
|
healthRefused.Add(1)
|
|
}
|
|
return err
|
|
}
|
|
if k == nil {
|
|
return nil
|
|
}
|
|
return judgeModuleHealth(ctx, inv, k, node, unhealthy, streaks, now)
|
|
}
|
|
|
|
// judgeModuleHealth raises a module's condition on a machine on the second statement in a row that says a
|
|
// resource of it is unhealthy — or on the first while it is already open — and clears every one this
|
|
// statement no longer says. **A consumer whose findings wait on an unhealthy provider is held** (to-be 48
|
|
// §6): raised as nothing of its own, listed at the provider's condition, which is urgent while anyone
|
|
// waits on it.
|
|
func judgeModuleHealth(ctx context.Context, inv *inventory.Inventory, k *conditions.Keeper, node string,
|
|
unhealthy map[string][]inventory.ResourceHealth, streaks map[string]int, now time.Time) error {
|
|
open, err := k.Open(ctx)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
standing := map[string]conditions.Condition{}
|
|
for _, c := range open {
|
|
if c.Kind == kindModuleUnhealthy && c.Subject.Machine == node {
|
|
standing[c.Key] = c
|
|
}
|
|
}
|
|
var hold *holding
|
|
if inv != nil {
|
|
if hold, err = readHolding(ctx, inv, open); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
var problems []string
|
|
modules := make([]string, 0, len(unhealthy))
|
|
for m := range unhealthy {
|
|
modules = append(modules, m)
|
|
}
|
|
sort.Strings(modules)
|
|
seen := map[string]bool{}
|
|
heldOn := map[string]string{}
|
|
providers := map[catalogue.Chosen]bool{}
|
|
for _, m := range modules {
|
|
o := moduleUnhealthyObservation(m, node, unhealthy[m])
|
|
if hold != nil {
|
|
if p, held := hold.heldUnder(node, m, unhealthy[m]); held {
|
|
// Held under the provider's condition: nothing of its own, and the provider's says it waits.
|
|
heldOn[o.Key()] = p.Module + " on " + p.Node
|
|
providers[p] = true
|
|
continue
|
|
}
|
|
if waiters := hold.waitersOn(catalogue.Chosen{Node: node, Module: m}); len(waiters) > 0 {
|
|
o.Severity = conditions.Urgent
|
|
o.Said += "; " + waitingWords(waiters)
|
|
o.Summary += fmt.Sprintf("; %d consumer(s) wait on it", len(waiters))
|
|
}
|
|
}
|
|
seen[o.Key()] = true
|
|
c, isOpen := standing[o.Key()]
|
|
if streaks[m] < moduleUnhealthyAfter && !isOpen {
|
|
continue // unconfirmed: one statement can be wrong; `node show` lists it
|
|
}
|
|
if isOpen && now.Sub(c.Raised) >= moduleUnhealthyUrgentAfter {
|
|
o.Severity = conditions.Urgent
|
|
}
|
|
if _, err := k.Observe(ctx, o); err != nil {
|
|
problems = append(problems, err.Error())
|
|
}
|
|
}
|
|
for key, c := range standing {
|
|
if seen[key] {
|
|
continue
|
|
}
|
|
module := strings.TrimSuffix(c.Subject.ID, "."+node)
|
|
why := fmt.Sprintf("%s says no resource of %s is unhealthy", node, module)
|
|
if on, held := heldOn[key]; held {
|
|
why = fmt.Sprintf("what %s finds on %s waits on %s, which is unhealthy: held under its condition", module, node, on)
|
|
}
|
|
if _, err := k.Clear(ctx, key, why); err != nil {
|
|
problems = append(problems, err.Error())
|
|
}
|
|
}
|
|
// And each provider a consumer here now waits on, when its own condition is open: said again with who
|
|
// waits on it, so the wait is listed at the provider whichever machine's statement arrived first.
|
|
for p := range providers {
|
|
if err := sayWaiters(ctx, k, hold, p, now); err != nil {
|
|
problems = append(problems, err.Error())
|
|
}
|
|
}
|
|
if len(problems) > 0 {
|
|
return fmt.Errorf("%s", strings.Join(problems, "; "))
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// sayWaiters observes a provider's open condition again, with who waits on it, from its machine's newest
|
|
// statement. Nothing when its condition is not open: it is raised by its own statements, on its own looks.
|
|
func sayWaiters(ctx context.Context, k *conditions.Keeper, hold *holding, p catalogue.Chosen, now time.Time) error {
|
|
var raisedAt *conditions.Condition
|
|
for i, c := range hold.open {
|
|
if c.Key == moduleUnhealthyKey(p.Module, p.Node) {
|
|
raisedAt = &hold.open[i]
|
|
}
|
|
}
|
|
if raisedAt == nil {
|
|
return nil
|
|
}
|
|
var rs []inventory.ResourceHealth
|
|
for _, r := range hold.healths[p.Node].Resources {
|
|
if r.Module == p.Module && r.State == link.StateUnhealthy {
|
|
rs = append(rs, r)
|
|
}
|
|
}
|
|
if len(rs) == 0 {
|
|
return nil
|
|
}
|
|
o := moduleUnhealthyObservation(p.Module, p.Node, rs)
|
|
if waiters := hold.waitersOn(p); len(waiters) > 0 {
|
|
o.Severity = conditions.Urgent
|
|
o.Said += "; " + waitingWords(waiters)
|
|
o.Summary += fmt.Sprintf("; %d consumer(s) wait on it", len(waiters))
|
|
}
|
|
_, err := k.Observe(ctx, o)
|
|
return err
|
|
}
|
|
|
|
// moduleUnhealthyObservation is a module unhealthy on a machine, in words: the summary names the module,
|
|
// the machine and what is wrong with each resource; the detail — targets, streaks, since — is evidence.
|
|
func moduleUnhealthyObservation(module, node string, rs []inventory.ResourceHealth) conditions.Observation {
|
|
var words, said []string
|
|
for _, r := range rs {
|
|
words = append(words, fmt.Sprintf("its %s %s %s", r.Kind, r.Resource, reasonWords(r)))
|
|
said = append(said, fmt.Sprintf("%s (%s %s): %s, %d look(s) in a row, %d restart(s) counted, since %s",
|
|
r.Resource, r.Kind, r.Target, orNotSaid(r.Reason), r.Streak, r.Restarts,
|
|
r.Since.UTC().Format("2006-01-02 15:04:05 MST")))
|
|
}
|
|
return conditions.Observation{Scope: conditions.ScopeModule, ID: module + "." + node, Token: "unhealthy",
|
|
Kind: kindModuleUnhealthy, Machine: node, Severity: conditions.Warning, Source: sourceHealth,
|
|
Summary: fmt.Sprintf("%s on %s is not healthy: %s", module, node, strings.Join(words, "; ")),
|
|
Said: strings.Join(said, "; ")}
|
|
}
|
|
|
|
// reasonWords is why a resource is unhealthy, as a person reads it.
|
|
func reasonWords(r inventory.ResourceHealth) string {
|
|
switch r.Reason {
|
|
case "restarting":
|
|
return fmt.Sprintf("keeps restarting (%d restart(s) counted)", r.Restarts)
|
|
case "down":
|
|
return "is not running"
|
|
case "":
|
|
return "is unhealthy"
|
|
}
|
|
// What a declared check found says an endpoint, a path or an address: evidence, never the summary the
|
|
// operator's channel carries (ADR 0234 §6). The summary names the check.
|
|
if r.Check != "" {
|
|
return "fails its " + r.Check + " check"
|
|
}
|
|
return "is unhealthy: " + r.Reason
|
|
}
|
|
|
|
func orNotSaid(s string) string {
|
|
if s == "" {
|
|
return "no reason said"
|
|
}
|
|
return s
|
|
}
|
|
|
|
// moduleHealthWord is the gate's reading of a module's stated health on a machine (ADR 0240 §4, ADR 0236
|
|
// §2 as amended): good when every long-running resource of it is stated healthy in a statement heard since
|
|
// the send; not yet otherwise, saying which. A machine that never stated health is judged as before.
|
|
func moduleHealthWord(module, machine string, since time.Time, f gateFacts) (health, string) {
|
|
if f.healthErr != nil {
|
|
return healthNotYet, "what " + machine + " says of its resources' health cannot be read: " + firstLine(f.healthErr.Error())
|
|
}
|
|
h, states := f.health[machine]
|
|
if !states {
|
|
return healthGood, ""
|
|
}
|
|
if h.HeardAt.Before(since) {
|
|
return healthNotYet, fmt.Sprintf("%s has not said how what %s runs is since it was sent", machine, module)
|
|
}
|
|
for _, r := range h.Resources {
|
|
if r.Module != module {
|
|
continue
|
|
}
|
|
switch r.State {
|
|
case link.StateHealthy:
|
|
case link.StateStarting:
|
|
return healthNotYet, fmt.Sprintf("its %s %s on %s is still starting", r.Kind, r.Resource, machine)
|
|
case link.StateUnhealthy:
|
|
if on, held := f.heldOn[module+"@"+machine]; held {
|
|
return healthWaiting, fmt.Sprintf("its %s %s on %s waits on %s, which is unhealthy", r.Kind,
|
|
r.Resource, machine, on)
|
|
}
|
|
return healthNotYet, fmt.Sprintf("its %s %s on %s %s", r.Kind, r.Resource, machine, reasonWords(r))
|
|
default:
|
|
return healthNotYet, fmt.Sprintf("its %s %s on %s is %s%s", r.Kind, r.Resource, machine, r.State,
|
|
reasonAfter(r.Reason))
|
|
}
|
|
}
|
|
return healthGood, ""
|
|
}
|
|
|
|
func reasonAfter(s string) string {
|
|
if s == "" {
|
|
return ""
|
|
}
|
|
return ": " + s
|
|
}
|
|
|
|
// healthLines is what `node show` says of a machine's long-running resources: each with its state and
|
|
// since when, an unhealthy one said once marked unconfirmed.
|
|
func healthLines(h inventory.NodeHealth, had bool, now time.Time) []string {
|
|
if !had {
|
|
return []string{" its node-engine does not say how what it runs is — it is older than the judging (ADR 0240)"}
|
|
}
|
|
if len(h.Resources) == 0 {
|
|
return []string{fmt.Sprintf(" it runs nothing long-lived for a module (said %s ago)", roughly(now.Sub(h.HeardAt)))}
|
|
}
|
|
out := []string{fmt.Sprintf(" what it runs, as it said %s ago:", roughly(now.Sub(h.HeardAt)))}
|
|
for _, r := range h.Resources {
|
|
line := fmt.Sprintf(" %-10s %-34s %s %s, since %s", r.State, r.Resource, r.Kind, r.Target,
|
|
r.Since.Local().Format("2006-01-02 15:04"))
|
|
if r.Reason != "" {
|
|
line += " — " + r.Reason
|
|
}
|
|
if r.Restarts > 0 {
|
|
line += fmt.Sprintf(", %d restart(s) counted", r.Restarts)
|
|
}
|
|
if r.State == link.StateUnhealthy && h.Streaks[r.Module] < moduleUnhealthyAfter {
|
|
line += " (unconfirmed: said once)"
|
|
}
|
|
out = append(out, line)
|
|
}
|
|
return out
|
|
}
|