04-ISSUES/010. The store now records where each resource came from -- carried, or declared -- and each origin removes only its own. A declaration removes what the mesh previously declared and never what the bundle raised. State written before the field existed reads as carried, because everything a host had applied by then came from its bundle: there was no other way to tell it anything. Guessing the other way would have the first upgrade remove the substrate, which is this fault arriving through the change that fixes it. Verified on the scenario that caused it, and on the property that had to survive it: a later declaration dropping a resource still removes that resource, so removal by omission still means what it meant. Also stops swallowing a publish failure. A node that applied a declaration and could not tell the mesh looked exactly like one that had -- the mesh believing it never answered, the node believing it did, and nothing anywhere saying so. Reports are published mandatory now, so anything the broker cannot route comes back and is said out loud rather than dropped in silence.
175 lines
6.4 KiB
Go
175 lines
6.4 KiB
Go
package link
|
|
|
|
import (
|
|
"context"
|
|
"crypto/ed25519"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"net/url"
|
|
"time"
|
|
|
|
amqp "github.com/rabbitmq/amqp091-go"
|
|
)
|
|
|
|
// ErrForged is what a node returns for a declaration whose signature is not the mesh's.
|
|
//
|
|
// Its own error, and it must never be confused with a malformed message. novox/hq ADR 0004
|
|
// requires a host to tell *this is not from the mesh I joined* apart from *this is malformed*:
|
|
// the first means somebody is trying, the second means something is broken.
|
|
var ErrForged = errors.New("this declaration was not signed by the mesh this node joined")
|
|
|
|
// Membership is what a node needs to reach its mesh again, held by the caller.
|
|
type Membership struct {
|
|
Node string
|
|
Broker string
|
|
Fingerprint string
|
|
Password string
|
|
Signer ed25519.PublicKey
|
|
}
|
|
|
|
// Applier is what the host does with a declaration that has been proved to come from the mesh.
|
|
type Applier func(ctx context.Context, declaration []byte) Report
|
|
|
|
// Announce is how the link says what is happening, so a node running unattended leaves an
|
|
// account of it. Nil is allowed and means say nothing.
|
|
type Announce func(string)
|
|
|
|
// Run holds the link open, applying what arrives and reporting what happened.
|
|
//
|
|
// Outbound only, and nothing listens on this machine. The connection is the node's presence in
|
|
// the mesh: while it is up the node is enrolled, and while it is down the node is disconnected —
|
|
// which is an ordinary situation and not a failure, so this returns rather than panicking and
|
|
// leaves restarting to whatever supervises it.
|
|
func Run(ctx context.Context, m Membership, apply Applier, say Announce, timeout time.Duration) error {
|
|
if say == nil {
|
|
say = func(string) {}
|
|
}
|
|
config, err := PinnedConfig(m.Fingerprint)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
dsn := fmt.Sprintf("amqps://%s:%s@%s/",
|
|
url.QueryEscape(m.Node), url.QueryEscape(m.Password), m.Broker)
|
|
conn, err := amqp.DialConfig(dsn, amqp.Config{
|
|
TLSClientConfig: config,
|
|
Dial: amqp.DefaultDial(timeout),
|
|
// Kept short so a node that has silently lost its route notices, rather than holding a
|
|
// connection the broker forgot about and believing it is still in the mesh.
|
|
Heartbeat: 10 * time.Second,
|
|
})
|
|
if err != nil {
|
|
if errors.Is(err, ErrWrongCertificate) {
|
|
return err
|
|
}
|
|
return fmt.Errorf("cannot reach the broker at %s: %w", m.Broker, err)
|
|
}
|
|
defer conn.Close()
|
|
|
|
channel, err := conn.Channel()
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer channel.Close()
|
|
|
|
queue := QueueFor(m.Node)
|
|
if _, err := channel.QueueDeclare(queue, true, false, false, false, nil); err != nil {
|
|
return fmt.Errorf("cannot declare this node's queue %s: %w", queue, err)
|
|
}
|
|
|
|
// One at a time. A declaration is applied to a machine, and applying two at once would race
|
|
// on the same filesystem — so the broker holds the next one until this one is finished,
|
|
// where it survives a restart.
|
|
if err := channel.Qos(1, 0, false); err != nil {
|
|
return err
|
|
}
|
|
|
|
deliveries, err := channel.ConsumeWithContext(ctx, queue, "", false, false, false, false, nil)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
closed := conn.NotifyClose(make(chan *amqp.Error, 1))
|
|
|
|
// Published mandatory, so the broker hands back anything it cannot route rather than
|
|
// dropping it. Without this a report goes to an exchange with no matching binding, the
|
|
// publisher is told nothing, and the mesh believes this node never answered while the node
|
|
// believes it did — which is what happened when `report` was left unbound on the other side.
|
|
returned := channel.NotifyReturn(make(chan amqp.Return, 4))
|
|
go func() {
|
|
for r := range returned {
|
|
say(fmt.Sprintf("the broker could not route this node's %s: %s (%d %s)",
|
|
r.RoutingKey, r.Exchange, r.ReplyCode, r.ReplyText))
|
|
}
|
|
}()
|
|
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return nil
|
|
case reason := <-closed:
|
|
return fmt.Errorf("the link closed: %v", reason)
|
|
case delivery, ok := <-deliveries:
|
|
if !ok {
|
|
return errors.New("the broker stopped delivering")
|
|
}
|
|
report := handle(ctx, m, apply, delivery)
|
|
switch {
|
|
case report.Refused != "":
|
|
say("refused a declaration: " + report.Refused)
|
|
case len(report.Failed) > 0:
|
|
say(fmt.Sprintf("applied %d and failed: %v", len(report.Applied), report.Failed))
|
|
default:
|
|
say(fmt.Sprintf("applied %d resource(s)", len(report.Applied)))
|
|
}
|
|
publishReport(ctx, channel, m, report, say, timeout)
|
|
// Acknowledged after the report is published. A node that dies between applying and
|
|
// reporting leaves the declaration on the broker and applies it again on return,
|
|
// which is safe because applying is reconciliation — it converges rather than
|
|
// repeating.
|
|
_ = delivery.Ack(false)
|
|
}
|
|
}
|
|
}
|
|
|
|
func handle(ctx context.Context, m Membership, apply Applier, delivery amqp.Delivery) Report {
|
|
return handleBody(ctx, m, delivery.Body, apply)
|
|
}
|
|
|
|
// handleBody is the whole of deciding whether to trust a message, separated from the broker so it
|
|
// can be tested as the security check it is rather than as message plumbing.
|
|
func handleBody(ctx context.Context, m Membership, body []byte, apply Applier) Report {
|
|
var signed Signed
|
|
if err := json.Unmarshal(body, &signed); err != nil {
|
|
return Report{Node: m.Node, Refused: "this message is not a declaration: " + err.Error()}
|
|
}
|
|
|
|
// Before anything is read out of it, let alone applied. The host applies whatever the link
|
|
// delivers, so this check is the difference between the mesh changing this machine and
|
|
// anybody changing it.
|
|
if !ed25519.Verify(m.Signer, signed.Declaration, signed.Signature) {
|
|
return Report{Node: m.Node, Refused: ErrForged.Error()}
|
|
}
|
|
return apply(ctx, signed.Declaration)
|
|
}
|
|
|
|
func publishReport(ctx context.Context, channel *amqp.Channel, m Membership, report Report,
|
|
say Announce, timeout time.Duration) {
|
|
report.Node = m.Node
|
|
body, err := json.Marshal(report)
|
|
if err != nil {
|
|
say("cannot encode this node's own report: " + err.Error())
|
|
return
|
|
}
|
|
publish, cancel := context.WithTimeout(ctx, timeout)
|
|
defer cancel()
|
|
|
|
// Said rather than swallowed. A report that fails to publish leaves the mesh believing this
|
|
// node never answered, while the node believes it did — and the two would go on disagreeing
|
|
// with nothing anywhere saying so. That shape of fault is the one this project keeps finding.
|
|
if err := channel.PublishWithContext(publish, Exchange, KeyReport, true, false,
|
|
amqp.Publishing{ContentType: "application/json", Body: body}); err != nil {
|
|
say(fmt.Sprintf("applied, and could not tell the mesh: %v", err))
|
|
}
|
|
}
|