A service can be declared to reflect a file
Because a running service does not re-read its configuration. Replace the file, find the service running, do nothing -- and the machine keeps behaving as it did while every check passes, because the file is right and the service is up. That is not hypothetical. It is how a third node joining a mesh left the first two carrying a private network that no longer existed, with every part of it reporting success. Declared state rather than a command: the declaration says the running service must reflect these files, and the host works out that it does not. A command to restart would be an action, and the link may not carry one -- the host refused precisely that when I tried it, correctly, which is how this shape was arrived at rather than the other. Scoped to one apply. A change from an earlier one has already been reflected, and restarting for it every time would make a steady machine bounce its services for ever. Also: the node generates its overlay key at enrolment and reports the public half, and the store waits three minutes rather than one for the database -- sixty seconds is not enough for a cold machine running initdb, and it failed that way three times, which is the worst kind of flake because a second run always fixed it.
This commit is contained in:
+51
-4
@@ -112,8 +112,13 @@ func Apply(
|
||||
log(fmt.Sprintf(" %s %s (%s)", action, orphan.ID, orphan.Target))
|
||||
}
|
||||
|
||||
// What moved in this apply, so a service that must reflect a file can be told the file
|
||||
// moved. Only within one apply: a change from an earlier one has already been reflected, and
|
||||
// restarting for it every time would make a steady machine restart its services for ever.
|
||||
changed := map[string]bool{}
|
||||
|
||||
for _, resource := range d.Resources {
|
||||
outcome, err := applyOne(ctx, sys, resource, run)
|
||||
outcome, err := applyOne(ctx, sys, resource, run, changed)
|
||||
if err != nil {
|
||||
return report, known, &Error{Resource: resource.Identity(), Err: err, Done: report}
|
||||
}
|
||||
@@ -126,20 +131,22 @@ func Apply(
|
||||
})
|
||||
report.Outcomes = append(report.Outcomes, outcome)
|
||||
if outcome.Action != "unchanged" {
|
||||
changed[resource.Identity()] = true
|
||||
log(fmt.Sprintf(" %s %s (%s)", outcome.Action, outcome.ID, outcome.Target))
|
||||
}
|
||||
}
|
||||
return report, known, nil
|
||||
}
|
||||
|
||||
func applyOne(ctx context.Context, sys system.System, r declaration.Resource, run Runner) (Outcome, error) {
|
||||
func applyOne(ctx context.Context, sys system.System, r declaration.Resource, run Runner,
|
||||
changed map[string]bool) (Outcome, error) {
|
||||
switch res := r.(type) {
|
||||
case *declaration.Directory:
|
||||
return applyDirectory(res)
|
||||
case *declaration.File:
|
||||
return applyFile(res)
|
||||
case *declaration.Service:
|
||||
return applyService(ctx, sys, res, run)
|
||||
return applyService(ctx, sys, res, run, changed)
|
||||
case *declaration.Package:
|
||||
return applyPackage(ctx, sys, res, run)
|
||||
case *declaration.Container:
|
||||
@@ -318,7 +325,25 @@ func writeAtomically(path string, content []byte, mode os.FileMode) error {
|
||||
return os.Rename(tmp.Name(), path)
|
||||
}
|
||||
|
||||
func applyService(ctx context.Context, sys system.System, r *declaration.Service, run Runner) (Outcome, error) {
|
||||
// reflects reports whether anything this service must mirror changed in this apply.
|
||||
func reflects(r *declaration.Service, changed map[string]bool) bool {
|
||||
return len(reflected(r, changed)) > 0
|
||||
}
|
||||
|
||||
// reflected is which of them changed, so the outcome can say why the service was restarted. A
|
||||
// restart with no reason given is indistinguishable from a service that keeps falling over.
|
||||
func reflected(r *declaration.Service, changed map[string]bool) []string {
|
||||
var which []string
|
||||
for _, id := range r.RestartOn {
|
||||
if changed[id] {
|
||||
which = append(which, id)
|
||||
}
|
||||
}
|
||||
return which
|
||||
}
|
||||
|
||||
func applyService(ctx context.Context, sys system.System, r *declaration.Service, run Runner,
|
||||
changed map[string]bool) (Outcome, error) {
|
||||
out := begin(r)
|
||||
var changes []string
|
||||
|
||||
@@ -365,6 +390,28 @@ func applyService(ctx context.Context, sys system.System, r *declaration.Service
|
||||
return out, fmt.Errorf("%s was asked to be %s and is %s", r.Unit, r.State, after)
|
||||
}
|
||||
changes = append(changes, before+" to "+after)
|
||||
} else if r.State == "running" && reflects(r, changed) {
|
||||
// The service is already in the state it was asked for, and something it must reflect
|
||||
// changed in this same apply. A running service does not re-read its configuration, so
|
||||
// leaving it alone here is how a machine ends up correct on disk and wrong in fact —
|
||||
// with every check passing.
|
||||
if err := sys.SetServiceState(ctx, run, r.Unit, "stopped"); err != nil {
|
||||
return out, fmt.Errorf("restarting %s: stopping it: %w", r.Unit, err)
|
||||
}
|
||||
if err := sys.SetServiceState(ctx, run, r.Unit, "running"); err != nil {
|
||||
return out, fmt.Errorf("restarting %s: starting it again: %w", r.Unit, err)
|
||||
}
|
||||
// Read back, for the same reason as above: a unit that starts and immediately dies
|
||||
// satisfies a service manager and nothing else.
|
||||
after, err := sys.ServiceState(ctx, run, r.Unit)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if after != "running" {
|
||||
return out, fmt.Errorf(
|
||||
"%s was restarted to pick up a change and is %s", r.Unit, after)
|
||||
}
|
||||
changes = append(changes, "restarted for "+strings.Join(reflected(r, changed), ", "))
|
||||
}
|
||||
|
||||
if len(changes) == 0 {
|
||||
|
||||
Reference in New Issue
Block a user