Measure each item its own way; never compare a partial size (hq ADR 0233)

A walk over a large library every hour loads the array that protects it. An item now says how it
is measured — a bounded daily walk, a dataset's counters, or its top level only — and a size that
is a lower bound is kept as such and never read as a shrink.
This commit is contained in:
jochen
2026-10-06 17:00:22 +02:00
parent 52af210e47
commit abf9125689
7 changed files with 156 additions and 19 deletions
+26 -1
View File
@@ -102,6 +102,10 @@ type DataItem struct {
Redundancy string `json:"redundancy,omitempty"`
// Within is how old its last good backup may be; unsaid, DefaultBackupWithin.
Within string `json:"within,omitempty"`
// Measure is how the backup holder measures it: `walk` (the default — every file, at most daily and
// bounded, for small items), `dataset` (a ZFS dataset's own counters, hourly, nothing walked) or
// `shallow` (its top-level entries only, no size). A large item never says walk.
Measure string `json:"measure,omitempty"`
// Active is how long it may go unwritten before that is a fault — for data something is
// expected to write all the time. Unsaid, a quiet item is not a fault.
Active string `json:"active,omitempty"`
@@ -207,6 +211,21 @@ func (it DataItem) Protection() string {
return strings.Join(by, "+")
}
// The ways an item is measured.
const (
MeasureWalk = "walk"
MeasureDataset = "dataset"
MeasureShallow = "shallow"
)
// MeasuredBy is how the item is measured, with the default applied.
func (it DataItem) MeasuredBy() string {
if it.Measure == "" {
return MeasureWalk
}
return it.Measure
}
// OwnedByModule is whether the item is in one of the module's own directories — the mesh's to retire —
// rather than an operator's path it was given, which the mesh never retires and never deletes.
func (it DataItem) OwnedByModule() bool { return strings.HasPrefix(it.Path, "${dir:") }
@@ -365,6 +384,11 @@ func (m Manifest) dataProblems() []string {
if it.Class == ClassCache && it.Redundancy != "" {
say("%s is a cache and says it is protected by redundancy; a cache is not protected", label)
}
switch it.Measure {
case "", MeasureWalk, MeasureDataset, MeasureShallow:
default:
say("%s's measure %q is walk, dataset or shallow", label, it.Measure)
}
if _, err := ParseDataDuration(it.Within); err != nil {
say("%s's within: %v", label, err)
}
@@ -541,7 +565,8 @@ func (m Manifest) derivedContributions() []SeatContribution {
case it.BackedUp():
covered = it.Path
}
described = append(described, fmt.Sprintf("item %s %s %s %s %s", it.ID, it.Class, it.Path, covered, it.Protection()))
described = append(described, fmt.Sprintf("item %s %s %s %s %s %s", it.ID, it.Class, it.Path, covered,
it.Protection(), it.MeasuredBy()))
}
var out []SeatContribution
if len(lines) > 0 {
+5 -4
View File
@@ -55,6 +55,7 @@ func TestTheDataSectionIsRefusedWhereItIsWrong(t *testing.T) {
{"irreplaceable and protected by nothing", `{"own":[{"id":"a","path":"${dir:d}","class":"irreplaceable","backup":"none"}]}`, "protected one way or the other"},
{"a cache on redundancy", `{"own":[{"id":"a","path":"${dir:d}","class":"cache","redundancy":"an array"}]}`, "a cache is not protected"},
{"kept with something not required", `{"kept-by":{"s3-bucket":{"class":"irreplaceable"}}}`, "does not require"},
{"a measure it does not know", `{"own":[{"id":"a","path":"${dir:d}","class":"cache","measure":"du"}]}`, "walk, dataset or shallow"},
{"a cache expecting writes", `{"own":[{"id":"a","path":"${dir:d}","class":"cache","active":"1d"}]}`, "only irreplaceable and valuable"},
{"a machine path", `{"own":[{"id":"a","path":"/srv/a","class":"cache"}]}`, "names no machine path"},
{"a directory it does not declare", `{"own":[{"id":"a","path":"${dir:nowhere}","class":"cache"}]}`, "declares no directory"},
@@ -168,7 +169,7 @@ func TestTheBackupHoldersLinesAreDerivedFromTheData(t *testing.T) {
"resources":[{"id":"home","type":"directory","path":"${machine:account-home}/.agent","mode":"0700"}]}`)
media := mustParse(t, `{"module":"media","version":"1",
"accesses":[{"id":"films","mode":"read"}],
"data":{"own":[{"id":"films","path":"${access:films}","class":"irreplaceable","redundancy":"on a redundant array; no room for a copy"},
"data":{"own":[{"id":"films","path":"${access:films}","class":"irreplaceable","redundancy":"on a redundant array; no room for a copy","measure":"dataset"},
{"id":"meta","path":"${dir:meta}","class":"rebuildable"}]},
"resources":[{"id":"meta","type":"directory","mode":"0700"}]}`)
facts := map[string]string{"account-home": "/home/op"}
@@ -185,9 +186,9 @@ func TestTheBackupHoldersLinesAreDerivedFromTheData(t *testing.T) {
if err != nil {
t.Fatal(err)
}
want = "# agent\nitem home valuable /home/op/.agent /home/op/.agent backup\nitem tmp cache /home/op/.agent/tmp - none\n" +
"# media\nitem films irreplaceable /tank/films - redundancy\nitem meta rebuildable /var/lib/media/meta /var/lib/media/meta backup\n" +
"# pg\nitem store irreplaceable /srv/store /srv/store/dumps backup\nitem dumps rebuildable /srv/store/dumps - none\n"
want = "# agent\nitem home valuable /home/op/.agent /home/op/.agent backup walk\nitem tmp cache /home/op/.agent/tmp - none walk\n" +
"# media\nitem films irreplaceable /tank/films - redundancy dataset\nitem meta rebuildable /var/lib/media/meta /var/lib/media/meta backup walk\n" +
"# pg\nitem store irreplaceable /srv/store /srv/store/dumps backup walk\nitem dumps rebuildable /srv/store/dumps - none walk\n"
if data != want {
t.Fatalf("data lines:\n%s\nwant:\n%s", data, want)
}
+30 -8
View File
@@ -4,6 +4,7 @@ import (
"context"
"errors"
"fmt"
"strings"
"time"
"github.com/jackc/pgx/v5"
@@ -43,7 +44,9 @@ type Measurement struct {
MeasuredAt *time.Time
LastBackup *time.Time
// Error is what the holder could not measure, when it could not.
Error string
Error string
// Precision is what the size is, as the holder said it: "exact", "dataset …", "partial: …", "no size: …".
Precision string
Redundancy *Redundancy
}
@@ -54,6 +57,7 @@ type DataRecord struct {
Protection string
FirstSeen, DeclaredAt time.Time
MeasureError string
Precision string
Redundancy *Redundancy
Size *int64
LastWrite, MeasuredAt, LastBackup *time.Time
@@ -63,6 +67,21 @@ type DataRecord struct {
DeletedBy, DeletedWhy string
}
// Comparable is whether a size is fit to compare with another: exact, or a dataset's own counters.
func Comparable(precision string) bool {
return precision == "" || precision == "exact" || strings.HasPrefix(precision, "dataset ")
}
// Dataset is the ZFS dataset a size was read from, when it was: what several items on one dataset share.
func Dataset(precision string) string {
rest, ok := strings.CutPrefix(precision, "dataset ")
if !ok {
return ""
}
name, _, _ := strings.Cut(rest, ":")
return name
}
// Retired is whether the item is retired and not deleted.
func (r DataRecord) Retired() bool { return r.RetiredAt != nil && r.DeletedAt == nil }
@@ -139,12 +158,15 @@ func (i *Inventory) RecordData(ctx context.Context, machine string, declared []D
if _, err := tx.Exec(ctx, `
insert into data_item (machine, module, item, class, path, first_seen, declared_at,
size_bytes, last_write, measured_at, last_backup, owned, protection,
measure_error, redundancy_kind, redundancy_where, redundancy_healthy, redundancy_said)
values ($1, $2, $3, $4, $5, $6, $6, $7, $8, $9, $10, $12, $13, $14, $15, $16, $17, $18)
measure_error, redundancy_kind, redundancy_where, redundancy_healthy, redundancy_said,
precision)
values ($1, $2, $3, $4, $5, $6, $6, $7, $8, $9, $10, $12, $13, $14, $15, $16, $17, $18, $19)
on conflict (machine, module, item) do update set
class = excluded.class,
owned = excluded.owned,
protection = excluded.protection,
precision = case when excluded.measured_at is not null then excluded.precision else data_item.precision end,
size_bytes = case when excluded.measured_at is not null then excluded.size_bytes else data_item.size_bytes end,
measure_error = case when excluded.measured_at is not null then excluded.measure_error else data_item.measure_error end,
redundancy_kind = case when excluded.measured_at is not null then excluded.redundancy_kind else data_item.redundancy_kind end,
redundancy_where = case when excluded.measured_at is not null then excluded.redundancy_where else data_item.redundancy_where end,
@@ -153,17 +175,17 @@ func (i *Inventory) RecordData(ctx context.Context, machine string, declared []D
path = case when excluded.path <> '' then excluded.path else data_item.path end,
first_seen = case when $11::boolean then excluded.first_seen else data_item.first_seen end,
declared_at = excluded.declared_at,
size_bytes = coalesce(excluded.size_bytes, data_item.size_bytes),
last_write = coalesce(excluded.last_write, data_item.last_write),
measured_at = coalesce(excluded.measured_at, data_item.measured_at),
last_backup = coalesce(excluded.last_backup, data_item.last_backup),
retired_at = null, retired_why = null,
deleted_at = null, deleted_by = null, deleted_why = null`,
machine, d.Module, d.Item, d.Class, m.Path, now, m.Size, m.LastWrite, m.MeasuredAt, m.LastBackup,
fresh, d.Owned, d.Protection, nullable(m.Error), red.Kind, red.Where, red.Healthy, red.Said); err != nil {
fresh, d.Owned, d.Protection, nullable(m.Error), red.Kind, red.Where, red.Healthy, red.Said,
nullable(m.Precision)); err != nil {
return change, err
}
if m.Size != nil && m.MeasuredAt != nil {
if m.Size != nil && m.MeasuredAt != nil && Comparable(m.Precision) {
if _, err := tx.Exec(ctx, `
insert into data_reading (machine, module, item, at, size_bytes, last_write)
select $1, $2, $3, $4, $5, $6
@@ -203,7 +225,7 @@ func (i *Inventory) RecordData(ctx context.Context, machine string, declared []D
const dataSelect = `select machine, module, item, class, path, first_seen, declared_at, size_bytes, last_write,
measured_at, last_backup, retired_at, coalesce(retired_why, ''), deleted_at, coalesce(deleted_by, ''),
coalesce(deleted_why, ''), owned, protection, coalesce(measure_error, ''), redundancy_kind, redundancy_where,
redundancy_healthy, redundancy_said from data_item`
redundancy_healthy, redundancy_said, coalesce(precision, '') from data_item`
func scanData(rows pgx.Row) (DataRecord, error) {
var r DataRecord
@@ -211,7 +233,7 @@ func scanData(rows pgx.Row) (DataRecord, error) {
var healthy *bool
err := rows.Scan(&r.Machine, &r.Module, &r.Item, &r.Class, &r.Path, &r.FirstSeen, &r.DeclaredAt, &r.Size,
&r.LastWrite, &r.MeasuredAt, &r.LastBackup, &r.RetiredAt, &r.RetiredWhy, &r.DeletedAt, &r.DeletedBy,
&r.DeletedWhy, &r.Owned, &r.Protection, &r.MeasureError, &kind, &where, &healthy, &said)
&r.DeletedWhy, &r.Owned, &r.Protection, &r.MeasureError, &kind, &where, &healthy, &said, &r.Precision)
if err == nil && kind != nil {
r.Redundancy = &Redundancy{Kind: *kind, Healthy: healthy}
if where != nil {
+21
View File
@@ -107,3 +107,24 @@ func TestReadingsAreHourlyAndThePeakIsTheLargest(t *testing.T) {
t.Fatalf("the newest measurement is not the one on record: %+v", r)
}
}
// A partial walk's lower bound is kept as the newest measurement, said as partial, and never kept as a
// reading a shrink would be read against.
func TestAPartialMeasurementIsNeverAReading(t *testing.T) {
inv := fresh(t)
ctx := t.Context()
now := time.Date(2026, 10, 6, 0, 0, 0, 0, time.UTC)
declared := []DeclaredData{{Module: "big", Item: "data", Class: "valuable", Owned: true}}
if _, err := inv.RecordData(ctx, "home", declared, map[string]map[string]Measurement{"big": {"data": {
Path: "/srv/big", Size: size(5), MeasuredAt: at(now), Precision: "partial: stopped"}}}, "", now); err != nil {
t.Fatal(err)
}
var n int
if err := inv.store.Pool().QueryRow(ctx, `select count(*) from data_reading`).Scan(&n); err != nil || n != 0 {
t.Fatalf("%d readings from a partial measurement: %v", n, err)
}
r, err := inv.DataOf(ctx, "home", "big", "data")
if err != nil || r.Precision != "partial: stopped" || Comparable(r.Precision) {
t.Fatalf("%+v, %v", r, err)
}
}
@@ -38,6 +38,8 @@ create table data_item (
last_backup timestamptz,
-- What the holder could not measure, when it could not: the path gone, a walk refused.
measure_error text,
-- What the newest size is: exact, a dataset's whole size, partial (a lower bound) or none.
precision text,
-- The redundant storage it is on, as the holder read it: zfs, md or btrfs, which pool or device,
-- whether it is healthy, and what it said.
redundancy_kind text,
@@ -53,7 +55,8 @@ create table data_item (
);
-- One row per measurement kept, at most one an hour per item, for ninety days: what a shrink is
-- read against.
-- read against. Only a comparable size is kept: exact, or a dataset's own counters — never a partial
-- walk's lower bound.
create table data_reading (
machine text not null,
module text not null,