Files
hq/00-META/checks/records.py
T
jschoubben b4607dfc03 Numbers are identity; the reading order is a generated, checked index
Decided after measuring what renumbering actually costs: 96 references in code
comments across two repositories, none of which would have failed to compile.
They would have pointed at the wrong reasoning, which is worse than a broken
link because nothing reports it.

So a number identifies a record and never changes. It cannot also be a
position -- a position moves when the set changes, and an identity that moves
is not one.

The reading order moves into an index generated from each record's `topic:`.
Six topics, in the order somebody learns the system.

The index is WRITTEN rather than only generated on demand, which reverses what
this repository previously said. The reason it said otherwise is that a
hand-written index drifts -- but a reader looking at the folder on a forge sees
the folder, not a command, and the drift objection is answered by checking
rather than by refusing to write one. That is §5's own rule: a rule states how
it is checked.

Two checks, both confirmed to bite. index.py fails when the written order no
longer matches the records. records.py fails when a record has no topic or one
nobody defined -- the quiet failure being a record that vanishes from the order
rather than appearing in the wrong place.
2026-08-28 23:39:18 +02:00

302 lines
12 KiB
Python

#!/usr/bin/env python3
"""Structural checks over HQ's own records.
Every check here exists because the thing it checks for actually happened. See README.md
for which incident is behind which check. Run from the repository root:
python3 00-META/checks/records.py
Exits non-zero if anything fails, so it can be a gate rather than a report.
"""
import os
import re
import sys
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
# Where a citation is *guidance* rather than history. A document here tells somebody what to
# do, so a link to a superseded record is an instruction to follow a withdrawn decision.
# 02-DECISIONS and 01-RESEARCH are deliberately absent: they record what was decided and what
# was observed, and both legitimately discuss superseded records at length.
GOVERNING = ("00-META/", "03-DESIGN/01-to-be/")
LINK = re.compile(r"\[[^\]]*\]\((?!https?:|mailto:)([^)]+)\)")
ADR_FILE = re.compile(r"^(\d{4})-")
def markdown_files():
for base, dirs, files in os.walk(ROOT):
dirs[:] = [d for d in dirs if d not in (".git", ".claude")]
for name in sorted(files):
if name.endswith(".md"):
yield os.path.join(base, name)
def rel(path):
return os.path.relpath(path, ROOT)
def read(path):
with open(path, encoding="utf-8") as handle:
return handle.read()
def frontmatter(text):
"""Minimal frontmatter reader — enough for the fields these checks use.
Not a YAML parser on purpose: a dependency in a repository that has none, to read four
scalar fields and one list, would cost more than it returns.
"""
if not text.startswith("---\n"):
return {}
end = text.find("\n---", 4)
if end == -1:
return {}
fields, key = {}, None
for line in text[4:end].split("\n"):
item = re.match(r"^\s+-\s+(.*)$", line)
if item and key:
fields.setdefault(key, []).append(item.group(1).strip())
continue
pair = re.match(r"^([A-Za-z_-]+):\s*(.*)$", line)
if pair:
key = pair.group(1)
value = pair.group(2).strip()
if value.startswith("[") and value.endswith("]"):
# inline list: `decisions: [a, b]`, `code: []`
inner = value[1:-1].strip()
fields[key] = [v.strip() for v in inner.split(",") if v.strip()]
elif value:
fields[key] = value
else:
fields[key] = []
return fields
def load_records():
"""Every decision record, by its four-digit number."""
records = {}
folder = os.path.join(ROOT, "02-DECISIONS")
for name in sorted(os.listdir(folder)):
match = ADR_FILE.match(name)
if not name.endswith(".md") or not match:
continue
path = os.path.join(folder, name)
text = read(path)
records[match.group(1)] = {
"number": match.group(1),
"name": name,
"path": path,
"text": text,
"front": frontmatter(text),
}
return records
class Failures:
def __init__(self):
self.items = []
def add(self, check, location, message):
self.items.append((check, location, message))
def report(self):
if not self.items:
print("records: all checks passed")
return 0
by_check = {}
for check, location, message in self.items:
by_check.setdefault(check, []).append((location, message))
for check in sorted(by_check):
print(f"\n{check} — {len(by_check[check])} problem(s)")
for location, message in by_check[check]:
print(f" {location}\n {message}")
print(f"\nrecords: {len(self.items)} problem(s)")
return 1
def check_links(failures):
"""Every relative link resolves to something that exists."""
for path in markdown_files():
folder = os.path.dirname(path)
for number, line in enumerate(read(path).split("\n"), 1):
for target in LINK.findall(line):
target = target.split("#")[0].strip()
if not target:
continue
if not os.path.exists(os.path.normpath(os.path.join(folder, target))):
failures.add("links", f"{rel(path)}:{number}", f"link does not resolve: {target}")
def check_rests_on(failures, records):
"""A document's `decisions:` and a record's `extends:` must name a live record.
These are the load-bearing citations: the document declares that it rests on that
decision. Resting on a withdrawn one is the defect this whole check set exists for.
"""
for path in markdown_files():
front = frontmatter(read(path))
cited = list(front.get("decisions", []) or [])
extends = front.get("extends")
if isinstance(extends, str) and extends:
cited.append(extends)
for entry in cited:
match = ADR_FILE.match(os.path.basename(entry))
if not match:
failures.add("rests-on", rel(path), f"not a decision record: {entry}")
continue
number = match.group(1)
if number not in records:
failures.add("rests-on", rel(path), f"no such record: {entry}")
continue
status = records[number]["front"].get("status")
if status != "accepted":
# A proposed record may extend another proposed one. Decisions are drafted in
# chains -- 0059 extends 0057 while both await review -- and refusing that would
# mean either drafting out of order or marking records accepted to satisfy a
# check, which is the failure this repository already made once.
if frontmatter(read(path)).get("status") == "proposed":
continue
# An as-is document describes what runs, and what runs was built under
# whatever was decided at the time. ADR 0056: "as-is describing a superseded
# decision is exactly what as-is is for."
if rel(path).startswith("03-DESIGN/00-as-is/"):
continue
# An extension that supersedes legitimately names what it replaced.
this = ADR_FILE.match(os.path.basename(path))
supersedes = records[number]["front"].get("superseded-by", "")
if this and supersedes and os.path.basename(path) in str(supersedes):
continue
failures.add(
"rests-on",
rel(path),
f"rests on ADR {number}, which is '{status}' — a document may not rest on a "
f"record that is not accepted",
)
def check_live_citations(failures, records):
"""In a governing document, a link to a superseded record must name its replacement.
The reader of a rule needs to know the rule was withdrawn, and needs somewhere to go.
Naming the superseder in the same paragraph is both, and it is what a person would
write anyway.
"""
superseded = {
number: record["front"].get("superseded-by", "")
for number, record in records.items()
if record["front"].get("status") == "superseded"
}
for path in markdown_files():
if not any(rel(path).startswith(prefix) for prefix in GOVERNING):
continue
text = read(path)
offset = 0
for paragraph in text.split("\n\n"):
line_no = text[:offset].count("\n") + 1
offset += len(paragraph) + 2
targets = LINK.findall(paragraph)
for target in targets:
match = ADR_FILE.match(os.path.basename(target.split("#")[0]))
if not match or match.group(1) not in superseded:
continue
number = match.group(1)
replacement = os.path.basename(str(superseded[number]))
if not replacement:
failures.add(
"live-citation",
f"{rel(path)}:{line_no}",
f"cites superseded ADR {number}, which names no superseder",
)
continue
if not any(replacement in t for t in targets):
failures.add(
"live-citation",
f"{rel(path)}:{line_no}",
f"cites superseded ADR {number} without naming its replacement "
f"({replacement}) in the same paragraph",
)
def check_supersession_symmetry(failures, records):
"""If A says it was superseded by B, B must say it supersedes A."""
for number, record in records.items():
front = record["front"]
status = front.get("status")
by = os.path.basename(str(front.get("superseded-by", "")))
if status == "superseded" and not by:
failures.add("supersession", rel(record["path"]), "marked superseded but names no superseder")
continue
if by and status != "superseded":
failures.add("supersession", rel(record["path"]), f"names a superseder but status is '{status}'")
if not by:
continue
match = ADR_FILE.match(by)
if not match or match.group(1) not in records:
failures.add("supersession", rel(record["path"]), f"superseder does not exist: {by}")
continue
other = records[match.group(1)]
claims = os.path.basename(str(other["front"].get("supersedes", "")))
if claims != record["name"]:
failures.add(
"supersession",
rel(other["path"]),
f"ADR {number} says this supersedes it; this record does not say so "
f"(supersedes: {claims or 'absent'})",
)
def check_topics(failures, records):
"""Every record names a topic the index knows.
The topic is what puts a record in the reading order, so a record without one — or with one
nobody defined — disappears from the index rather than appearing in the wrong place. That is
the quiet failure, so it is the one checked.
"""
known = {"the mesh", "the tiers", "what runs on it", "building it", "checking it",
"how we work"}
for number, record in sorted(records.items()):
topic = record["front"].get("topic")
if not topic:
failures.add("topics", rel(record["path"]),
"no topic, so it has no place in the reading order")
elif topic not in known:
failures.add("topics", rel(record["path"]),
"topic %r is not one of: %s" % (topic, ", ".join(sorted(known))))
def check_numbering(failures, records):
"""The number in the filename is the number in the heading."""
for number, record in records.items():
heading = re.search(r"^# (\d+)\.", record["text"], re.M)
if not heading:
failures.add("numbering", rel(record["path"]), "no '# N. Title' heading")
elif heading.group(1) != number.lstrip("0"):
failures.add(
"numbering",
rel(record["path"]),
f"filename says {number}, heading says {heading.group(1)}",
)
def main():
failures = Failures()
records = load_records()
check_links(failures)
check_rests_on(failures, records)
check_live_citations(failures, records)
check_supersession_symmetry(failures, records)
check_numbering(failures, records)
check_topics(failures, records)
print(f"records: {len(records)} decision records checked")
return failures.report()
if __name__ == "__main__":
sys.exit(main())