A provider now waits for a person before retiring more than three consumers or half of what it holds, and deletes only when asked. The controller is that person's way in: it keeps waiting and rejected sets as conditions, answers them with retire approve|reject, lists and deletes retired consumers through the provider's own tools on its machine, records each act in the hand-act log, and probes for anything retired longer than thirty days (D11).
190 lines
6.8 KiB
Go
190 lines
6.8 KiB
Go
package main
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/novox/mesh-controller/internal/conditions"
|
|
"github.com/novox/mesh-controller/internal/inventory"
|
|
"github.com/novox/mesh-controller/internal/link"
|
|
)
|
|
|
|
// A provider that keeps failing a consumer is a problem the controller reports (novox/hq ADR 0224) —
|
|
// **the condition store's first kind** (to-be 45 §2, ADR 0227).
|
|
//
|
|
// On 2026-10-05 the identity provider's provisioner failed every consumer from shortly after midnight
|
|
// until it was fixed by hand that night — 31,000 refused logins after its database was moved and its
|
|
// admin kept an older password — and `status` called the mesh well all day (novox/hq issue 179). A
|
|
// provider announces a consumer it has failed for minutes; the controller keeps it until the provider
|
|
// says it recovered; and `status`, its JSON and `node show` name it, breaking "all well". Unchanged in
|
|
// what it says and when; kept as a condition, `provider.<module>.<node>.<consumer>.failing`, rather
|
|
// than a row of its own, so it is said outward like every other fault and silenced like one.
|
|
|
|
// Kinds of the provider standing.
|
|
const (
|
|
kindProviderFailing = "provider-failing"
|
|
kindProviderSilent = "provider-silent"
|
|
// sourceProvisioner is what raised a standing: the provider's own event.
|
|
sourceProvisioner = "provisioner.failing"
|
|
)
|
|
|
|
// providerSaysAgainWithin is how long a failing word stays current without being said again: twice
|
|
// the quarter of an hour a provider repeats it at (ADR 0224). Past it, S8.
|
|
const providerSaysAgainWithin = 30 * time.Minute
|
|
|
|
// standings keeps what providers say, as conditions.
|
|
type standings struct {
|
|
keeper func() *conditions.Keeper
|
|
}
|
|
|
|
// standingObservation is a provider's failing word as an observation: the provider, its machine and
|
|
// the consumer name it, so the same consumer failed again is the same condition.
|
|
func standingObservation(st link.Standing) conditions.Observation {
|
|
whom := st.Consumer
|
|
if st.Node != "" {
|
|
whom += " on " + st.Node
|
|
}
|
|
summary := fmt.Sprintf("%s on %s keeps failing %s: %s, %d attempt(s) since %s", st.Module, st.ProviderNode,
|
|
whom, orUnclassed(st.Class), st.Attempts, st.Since.UTC().Format("2006-01-02 15:04 MST"))
|
|
said := orUnclassed(st.Class)
|
|
if e := firstLine(st.Error); e != "" {
|
|
said += ": " + e
|
|
}
|
|
if st.Provider != "" {
|
|
said += fmt.Sprintf(" (provision %s, %d attempts)", st.Provider, st.Attempts)
|
|
}
|
|
return conditions.Observation{Scope: conditions.ScopeProvider,
|
|
ID: st.Module + "." + st.ProviderNode + "." + st.Consumer,
|
|
Token: "failing", Kind: kindProviderFailing, Machine: st.ProviderNode, Also: alsoOn(st.Node, st.ProviderNode),
|
|
Severity: conditions.Warning, Summary: summary, Said: said, Source: sourceProvisioner}
|
|
}
|
|
|
|
// Stood keeps a provider's newest word: failing raises or observes its condition, recovered clears
|
|
// it. An error is the store away, and the link holds the message to be asked again — a recovery is
|
|
// said once, and dropping it would leave a consumer named failing that is fine.
|
|
func (s standings) Stood(ctx context.Context, st link.Standing) (bool, error) {
|
|
k := s.keeper()
|
|
if k == nil {
|
|
return false, fmt.Errorf("the condition store is not open in this controller: %w", link.ErrTryAgain)
|
|
}
|
|
o := standingObservation(st)
|
|
if !st.Failing {
|
|
why := "the provider says it recovered"
|
|
if st.Why != "" {
|
|
why += ": " + st.Why
|
|
}
|
|
cleared, err := k.Clear(ctx, o.Key(), why)
|
|
return cleared, storeAway(err)
|
|
}
|
|
_, err := k.Observe(ctx, o)
|
|
return false, storeAway(err)
|
|
}
|
|
|
|
// storeAway reads the condition store failing as the bus being away for the moment: the link holds the
|
|
// message and asks again, as it does for a store restarting (ADR 0083), rather than taking it unkept.
|
|
func storeAway(err error) error {
|
|
if err == nil {
|
|
return nil
|
|
}
|
|
return fmt.Errorf("%v: %w", err, link.ErrTryAgain)
|
|
}
|
|
|
|
// providerStandings is every open provider-failing condition, from what is open.
|
|
func providerStandings(open []conditions.Condition) []conditions.Condition {
|
|
var out []conditions.Condition
|
|
for _, c := range open {
|
|
if c.Kind == kindProviderFailing {
|
|
out = append(out, c)
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// providerConditions is every open condition a provider's word raised: a consumer failing (ADR 0224),
|
|
// and a retirement waiting for a person, kept by a rejection, or cleanup waiting (ADR 0230).
|
|
func providerConditions(open []conditions.Condition) []conditions.Condition {
|
|
var out []conditions.Condition
|
|
for _, c := range open {
|
|
switch c.Kind {
|
|
case kindProviderFailing, kindRetireWaiting, kindRetireRejected, kindCleanupWaiting:
|
|
out = append(out, c)
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// providerOf reads a provider condition's module and machine back from its key: a standing's
|
|
// `provider.<module>.<node>.<consumer>.failing`, and a retirement's `provider.<module>.<node>.<token>`,
|
|
// which names no consumer.
|
|
func providerOf(c conditions.Condition) (module, node, consumer string, ok bool) {
|
|
parts := strings.Split(c.Key, ".")
|
|
if parts[0] != conditions.ScopeProvider {
|
|
return "", "", "", false
|
|
}
|
|
switch len(parts) {
|
|
case 5:
|
|
return parts[1], parts[2], parts[3], true
|
|
case 4:
|
|
return parts[1], parts[2], "", true
|
|
}
|
|
return "", "", "", false
|
|
}
|
|
|
|
// unassignedProviders clears the standing of every provider no longer assigned where it ran — and its
|
|
// retirement and cleanup conditions with it (ADR 0230): nothing runs there to retire or delete anything.
|
|
//
|
|
// **A provider no longer assigned is not asked about** (ADR 0224 §4): nothing runs there to fail
|
|
// anybody, and nothing there will ever say it recovered. The observation that resolves it is the
|
|
// assignment. Assigned again, its first failure raises it again.
|
|
func unassignedProviders(ctx context.Context, inv *inventory.Inventory, k *conditions.Keeper,
|
|
open []conditions.Condition) error {
|
|
assigned := map[string]map[string]bool{}
|
|
for _, c := range providerConditions(open) {
|
|
module, node, _, ok := providerOf(c)
|
|
if !ok {
|
|
continue
|
|
}
|
|
on, asked := assigned[node]
|
|
if !asked {
|
|
modules, err := inv.Assigned(ctx, node)
|
|
if err != nil {
|
|
if errors.Is(err, inventory.ErrNoSuchNode) {
|
|
modules = nil // a machine the mesh no longer knows runs nothing
|
|
} else {
|
|
return fmt.Errorf("what %s is assigned cannot be read: %w", node, err)
|
|
}
|
|
}
|
|
on = map[string]bool{}
|
|
for _, m := range modules {
|
|
on[m] = true
|
|
}
|
|
assigned[node] = on
|
|
}
|
|
if !on[module] {
|
|
if _, err := k.Clear(ctx, c.Key, module+" is no longer assigned to "+node+
|
|
": nothing runs there to fail anybody"); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func orUnclassed(class string) string {
|
|
if class == "" {
|
|
return "failing"
|
|
}
|
|
return class
|
|
}
|
|
|
|
// alsoOn is a consumer's machine, when it is not the provider's.
|
|
func alsoOn(consumerNode, providerNode string) []string {
|
|
if consumerNode == "" || consumerNode == providerNode {
|
|
return nil
|
|
}
|
|
return []string{consumerNode}
|
|
}
|