Files
mesh-controller/cmd/mesh-controller/standing.go
T
jochen 68009b16fe Hear what providers retire, and let a person approve, reject and delete (hq ADR 0230)
A provider now waits for a person before retiring more than three consumers
or half of what it holds, and deletes only when asked. The controller is that
person's way in: it keeps waiting and rejected sets as conditions, answers them
with retire approve|reject, lists and deletes retired consumers through the
provider's own tools on its machine, records each act in the hand-act log, and
probes for anything retired longer than thirty days (D11).
2026-10-06 14:41:15 +02:00

190 lines
6.8 KiB
Go

package main
import (
"context"
"errors"
"fmt"
"strings"
"time"
"github.com/novox/mesh-controller/internal/conditions"
"github.com/novox/mesh-controller/internal/inventory"
"github.com/novox/mesh-controller/internal/link"
)
// A provider that keeps failing a consumer is a problem the controller reports (novox/hq ADR 0224) —
// **the condition store's first kind** (to-be 45 §2, ADR 0227).
//
// On 2026-10-05 the identity provider's provisioner failed every consumer from shortly after midnight
// until it was fixed by hand that night — 31,000 refused logins after its database was moved and its
// admin kept an older password — and `status` called the mesh well all day (novox/hq issue 179). A
// provider announces a consumer it has failed for minutes; the controller keeps it until the provider
// says it recovered; and `status`, its JSON and `node show` name it, breaking "all well". Unchanged in
// what it says and when; kept as a condition, `provider.<module>.<node>.<consumer>.failing`, rather
// than a row of its own, so it is said outward like every other fault and silenced like one.
// Kinds of the provider standing.
const (
kindProviderFailing = "provider-failing"
kindProviderSilent = "provider-silent"
// sourceProvisioner is what raised a standing: the provider's own event.
sourceProvisioner = "provisioner.failing"
)
// providerSaysAgainWithin is how long a failing word stays current without being said again: twice
// the quarter of an hour a provider repeats it at (ADR 0224). Past it, S8.
const providerSaysAgainWithin = 30 * time.Minute
// standings keeps what providers say, as conditions.
type standings struct {
keeper func() *conditions.Keeper
}
// standingObservation is a provider's failing word as an observation: the provider, its machine and
// the consumer name it, so the same consumer failed again is the same condition.
func standingObservation(st link.Standing) conditions.Observation {
whom := st.Consumer
if st.Node != "" {
whom += " on " + st.Node
}
summary := fmt.Sprintf("%s on %s keeps failing %s: %s, %d attempt(s) since %s", st.Module, st.ProviderNode,
whom, orUnclassed(st.Class), st.Attempts, st.Since.UTC().Format("2006-01-02 15:04 MST"))
said := orUnclassed(st.Class)
if e := firstLine(st.Error); e != "" {
said += ": " + e
}
if st.Provider != "" {
said += fmt.Sprintf(" (provision %s, %d attempts)", st.Provider, st.Attempts)
}
return conditions.Observation{Scope: conditions.ScopeProvider,
ID: st.Module + "." + st.ProviderNode + "." + st.Consumer,
Token: "failing", Kind: kindProviderFailing, Machine: st.ProviderNode, Also: alsoOn(st.Node, st.ProviderNode),
Severity: conditions.Warning, Summary: summary, Said: said, Source: sourceProvisioner}
}
// Stood keeps a provider's newest word: failing raises or observes its condition, recovered clears
// it. An error is the store away, and the link holds the message to be asked again — a recovery is
// said once, and dropping it would leave a consumer named failing that is fine.
func (s standings) Stood(ctx context.Context, st link.Standing) (bool, error) {
k := s.keeper()
if k == nil {
return false, fmt.Errorf("the condition store is not open in this controller: %w", link.ErrTryAgain)
}
o := standingObservation(st)
if !st.Failing {
why := "the provider says it recovered"
if st.Why != "" {
why += ": " + st.Why
}
cleared, err := k.Clear(ctx, o.Key(), why)
return cleared, storeAway(err)
}
_, err := k.Observe(ctx, o)
return false, storeAway(err)
}
// storeAway reads the condition store failing as the bus being away for the moment: the link holds the
// message and asks again, as it does for a store restarting (ADR 0083), rather than taking it unkept.
func storeAway(err error) error {
if err == nil {
return nil
}
return fmt.Errorf("%v: %w", err, link.ErrTryAgain)
}
// providerStandings is every open provider-failing condition, from what is open.
func providerStandings(open []conditions.Condition) []conditions.Condition {
var out []conditions.Condition
for _, c := range open {
if c.Kind == kindProviderFailing {
out = append(out, c)
}
}
return out
}
// providerConditions is every open condition a provider's word raised: a consumer failing (ADR 0224),
// and a retirement waiting for a person, kept by a rejection, or cleanup waiting (ADR 0230).
func providerConditions(open []conditions.Condition) []conditions.Condition {
var out []conditions.Condition
for _, c := range open {
switch c.Kind {
case kindProviderFailing, kindRetireWaiting, kindRetireRejected, kindCleanupWaiting:
out = append(out, c)
}
}
return out
}
// providerOf reads a provider condition's module and machine back from its key: a standing's
// `provider.<module>.<node>.<consumer>.failing`, and a retirement's `provider.<module>.<node>.<token>`,
// which names no consumer.
func providerOf(c conditions.Condition) (module, node, consumer string, ok bool) {
parts := strings.Split(c.Key, ".")
if parts[0] != conditions.ScopeProvider {
return "", "", "", false
}
switch len(parts) {
case 5:
return parts[1], parts[2], parts[3], true
case 4:
return parts[1], parts[2], "", true
}
return "", "", "", false
}
// unassignedProviders clears the standing of every provider no longer assigned where it ran — and its
// retirement and cleanup conditions with it (ADR 0230): nothing runs there to retire or delete anything.
//
// **A provider no longer assigned is not asked about** (ADR 0224 §4): nothing runs there to fail
// anybody, and nothing there will ever say it recovered. The observation that resolves it is the
// assignment. Assigned again, its first failure raises it again.
func unassignedProviders(ctx context.Context, inv *inventory.Inventory, k *conditions.Keeper,
open []conditions.Condition) error {
assigned := map[string]map[string]bool{}
for _, c := range providerConditions(open) {
module, node, _, ok := providerOf(c)
if !ok {
continue
}
on, asked := assigned[node]
if !asked {
modules, err := inv.Assigned(ctx, node)
if err != nil {
if errors.Is(err, inventory.ErrNoSuchNode) {
modules = nil // a machine the mesh no longer knows runs nothing
} else {
return fmt.Errorf("what %s is assigned cannot be read: %w", node, err)
}
}
on = map[string]bool{}
for _, m := range modules {
on[m] = true
}
assigned[node] = on
}
if !on[module] {
if _, err := k.Clear(ctx, c.Key, module+" is no longer assigned to "+node+
": nothing runs there to fail anybody"); err != nil {
return err
}
}
}
return nil
}
func orUnclassed(class string) string {
if class == "" {
return "failing"
}
return class
}
// alsoOn is a consumer's machine, when it is not the provider's.
func alsoOn(consumerNode, providerNode string) []string {
if consumerNode == "" || consumerNode == providerNode {
return nil
}
return []string{consumerNode}
}