Decided after measuring what renumbering actually costs: 96 references in code comments across two repositories, none of which would have failed to compile. They would have pointed at the wrong reasoning, which is worse than a broken link because nothing reports it. So a number identifies a record and never changes. It cannot also be a position -- a position moves when the set changes, and an identity that moves is not one. The reading order moves into an index generated from each record's `topic:`. Six topics, in the order somebody learns the system. The index is WRITTEN rather than only generated on demand, which reverses what this repository previously said. The reason it said otherwise is that a hand-written index drifts -- but a reader looking at the folder on a forge sees the folder, not a command, and the drift objection is answered by checking rather than by refusing to write one. That is §5's own rule: a rule states how it is checked. Two checks, both confirmed to bite. index.py fails when the written order no longer matches the records. records.py fails when a record has no topic or one nobody defined -- the quiet failure being a record that vanishes from the order rather than appearing in the wrong place.
302 lines
12 KiB
Python
302 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""Structural checks over HQ's own records.
|
|
|
|
Every check here exists because the thing it checks for actually happened. See README.md
|
|
for which incident is behind which check. Run from the repository root:
|
|
|
|
python3 00-META/checks/records.py
|
|
|
|
Exits non-zero if anything fails, so it can be a gate rather than a report.
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
|
|
# Where a citation is *guidance* rather than history. A document here tells somebody what to
|
|
# do, so a link to a superseded record is an instruction to follow a withdrawn decision.
|
|
# 02-DECISIONS and 01-RESEARCH are deliberately absent: they record what was decided and what
|
|
# was observed, and both legitimately discuss superseded records at length.
|
|
GOVERNING = ("00-META/", "03-DESIGN/01-to-be/")
|
|
|
|
LINK = re.compile(r"\[[^\]]*\]\((?!https?:|mailto:)([^)]+)\)")
|
|
ADR_FILE = re.compile(r"^(\d{4})-")
|
|
|
|
|
|
def markdown_files():
|
|
for base, dirs, files in os.walk(ROOT):
|
|
dirs[:] = [d for d in dirs if d not in (".git", ".claude")]
|
|
for name in sorted(files):
|
|
if name.endswith(".md"):
|
|
yield os.path.join(base, name)
|
|
|
|
|
|
def rel(path):
|
|
return os.path.relpath(path, ROOT)
|
|
|
|
|
|
def read(path):
|
|
with open(path, encoding="utf-8") as handle:
|
|
return handle.read()
|
|
|
|
|
|
def frontmatter(text):
|
|
"""Minimal frontmatter reader — enough for the fields these checks use.
|
|
|
|
Not a YAML parser on purpose: a dependency in a repository that has none, to read four
|
|
scalar fields and one list, would cost more than it returns.
|
|
"""
|
|
if not text.startswith("---\n"):
|
|
return {}
|
|
end = text.find("\n---", 4)
|
|
if end == -1:
|
|
return {}
|
|
fields, key = {}, None
|
|
for line in text[4:end].split("\n"):
|
|
item = re.match(r"^\s+-\s+(.*)$", line)
|
|
if item and key:
|
|
fields.setdefault(key, []).append(item.group(1).strip())
|
|
continue
|
|
pair = re.match(r"^([A-Za-z_-]+):\s*(.*)$", line)
|
|
if pair:
|
|
key = pair.group(1)
|
|
value = pair.group(2).strip()
|
|
if value.startswith("[") and value.endswith("]"):
|
|
# inline list: `decisions: [a, b]`, `code: []`
|
|
inner = value[1:-1].strip()
|
|
fields[key] = [v.strip() for v in inner.split(",") if v.strip()]
|
|
elif value:
|
|
fields[key] = value
|
|
else:
|
|
fields[key] = []
|
|
return fields
|
|
|
|
|
|
def load_records():
|
|
"""Every decision record, by its four-digit number."""
|
|
records = {}
|
|
folder = os.path.join(ROOT, "02-DECISIONS")
|
|
for name in sorted(os.listdir(folder)):
|
|
match = ADR_FILE.match(name)
|
|
if not name.endswith(".md") or not match:
|
|
continue
|
|
path = os.path.join(folder, name)
|
|
text = read(path)
|
|
records[match.group(1)] = {
|
|
"number": match.group(1),
|
|
"name": name,
|
|
"path": path,
|
|
"text": text,
|
|
"front": frontmatter(text),
|
|
}
|
|
return records
|
|
|
|
|
|
class Failures:
|
|
def __init__(self):
|
|
self.items = []
|
|
|
|
def add(self, check, location, message):
|
|
self.items.append((check, location, message))
|
|
|
|
def report(self):
|
|
if not self.items:
|
|
print("records: all checks passed")
|
|
return 0
|
|
by_check = {}
|
|
for check, location, message in self.items:
|
|
by_check.setdefault(check, []).append((location, message))
|
|
for check in sorted(by_check):
|
|
print(f"\n{check} — {len(by_check[check])} problem(s)")
|
|
for location, message in by_check[check]:
|
|
print(f" {location}\n {message}")
|
|
print(f"\nrecords: {len(self.items)} problem(s)")
|
|
return 1
|
|
|
|
|
|
def check_links(failures):
|
|
"""Every relative link resolves to something that exists."""
|
|
for path in markdown_files():
|
|
folder = os.path.dirname(path)
|
|
for number, line in enumerate(read(path).split("\n"), 1):
|
|
for target in LINK.findall(line):
|
|
target = target.split("#")[0].strip()
|
|
if not target:
|
|
continue
|
|
if not os.path.exists(os.path.normpath(os.path.join(folder, target))):
|
|
failures.add("links", f"{rel(path)}:{number}", f"link does not resolve: {target}")
|
|
|
|
|
|
def check_rests_on(failures, records):
|
|
"""A document's `decisions:` and a record's `extends:` must name a live record.
|
|
|
|
These are the load-bearing citations: the document declares that it rests on that
|
|
decision. Resting on a withdrawn one is the defect this whole check set exists for.
|
|
"""
|
|
for path in markdown_files():
|
|
front = frontmatter(read(path))
|
|
cited = list(front.get("decisions", []) or [])
|
|
extends = front.get("extends")
|
|
if isinstance(extends, str) and extends:
|
|
cited.append(extends)
|
|
|
|
for entry in cited:
|
|
match = ADR_FILE.match(os.path.basename(entry))
|
|
if not match:
|
|
failures.add("rests-on", rel(path), f"not a decision record: {entry}")
|
|
continue
|
|
number = match.group(1)
|
|
if number not in records:
|
|
failures.add("rests-on", rel(path), f"no such record: {entry}")
|
|
continue
|
|
status = records[number]["front"].get("status")
|
|
if status != "accepted":
|
|
# A proposed record may extend another proposed one. Decisions are drafted in
|
|
# chains -- 0059 extends 0057 while both await review -- and refusing that would
|
|
# mean either drafting out of order or marking records accepted to satisfy a
|
|
# check, which is the failure this repository already made once.
|
|
if frontmatter(read(path)).get("status") == "proposed":
|
|
continue
|
|
# An as-is document describes what runs, and what runs was built under
|
|
# whatever was decided at the time. ADR 0056: "as-is describing a superseded
|
|
# decision is exactly what as-is is for."
|
|
if rel(path).startswith("03-DESIGN/00-as-is/"):
|
|
continue
|
|
# An extension that supersedes legitimately names what it replaced.
|
|
this = ADR_FILE.match(os.path.basename(path))
|
|
supersedes = records[number]["front"].get("superseded-by", "")
|
|
if this and supersedes and os.path.basename(path) in str(supersedes):
|
|
continue
|
|
failures.add(
|
|
"rests-on",
|
|
rel(path),
|
|
f"rests on ADR {number}, which is '{status}' — a document may not rest on a "
|
|
f"record that is not accepted",
|
|
)
|
|
|
|
|
|
def check_live_citations(failures, records):
|
|
"""In a governing document, a link to a superseded record must name its replacement.
|
|
|
|
The reader of a rule needs to know the rule was withdrawn, and needs somewhere to go.
|
|
Naming the superseder in the same paragraph is both, and it is what a person would
|
|
write anyway.
|
|
"""
|
|
superseded = {
|
|
number: record["front"].get("superseded-by", "")
|
|
for number, record in records.items()
|
|
if record["front"].get("status") == "superseded"
|
|
}
|
|
|
|
for path in markdown_files():
|
|
if not any(rel(path).startswith(prefix) for prefix in GOVERNING):
|
|
continue
|
|
text = read(path)
|
|
offset = 0
|
|
for paragraph in text.split("\n\n"):
|
|
line_no = text[:offset].count("\n") + 1
|
|
offset += len(paragraph) + 2
|
|
targets = LINK.findall(paragraph)
|
|
for target in targets:
|
|
match = ADR_FILE.match(os.path.basename(target.split("#")[0]))
|
|
if not match or match.group(1) not in superseded:
|
|
continue
|
|
number = match.group(1)
|
|
replacement = os.path.basename(str(superseded[number]))
|
|
if not replacement:
|
|
failures.add(
|
|
"live-citation",
|
|
f"{rel(path)}:{line_no}",
|
|
f"cites superseded ADR {number}, which names no superseder",
|
|
)
|
|
continue
|
|
if not any(replacement in t for t in targets):
|
|
failures.add(
|
|
"live-citation",
|
|
f"{rel(path)}:{line_no}",
|
|
f"cites superseded ADR {number} without naming its replacement "
|
|
f"({replacement}) in the same paragraph",
|
|
)
|
|
|
|
|
|
def check_supersession_symmetry(failures, records):
|
|
"""If A says it was superseded by B, B must say it supersedes A."""
|
|
for number, record in records.items():
|
|
front = record["front"]
|
|
status = front.get("status")
|
|
by = os.path.basename(str(front.get("superseded-by", "")))
|
|
|
|
if status == "superseded" and not by:
|
|
failures.add("supersession", rel(record["path"]), "marked superseded but names no superseder")
|
|
continue
|
|
if by and status != "superseded":
|
|
failures.add("supersession", rel(record["path"]), f"names a superseder but status is '{status}'")
|
|
if not by:
|
|
continue
|
|
|
|
match = ADR_FILE.match(by)
|
|
if not match or match.group(1) not in records:
|
|
failures.add("supersession", rel(record["path"]), f"superseder does not exist: {by}")
|
|
continue
|
|
other = records[match.group(1)]
|
|
claims = os.path.basename(str(other["front"].get("supersedes", "")))
|
|
if claims != record["name"]:
|
|
failures.add(
|
|
"supersession",
|
|
rel(other["path"]),
|
|
f"ADR {number} says this supersedes it; this record does not say so "
|
|
f"(supersedes: {claims or 'absent'})",
|
|
)
|
|
|
|
|
|
def check_topics(failures, records):
|
|
"""Every record names a topic the index knows.
|
|
|
|
The topic is what puts a record in the reading order, so a record without one — or with one
|
|
nobody defined — disappears from the index rather than appearing in the wrong place. That is
|
|
the quiet failure, so it is the one checked.
|
|
"""
|
|
known = {"the mesh", "the tiers", "what runs on it", "building it", "checking it",
|
|
"how we work"}
|
|
for number, record in sorted(records.items()):
|
|
topic = record["front"].get("topic")
|
|
if not topic:
|
|
failures.add("topics", rel(record["path"]),
|
|
"no topic, so it has no place in the reading order")
|
|
elif topic not in known:
|
|
failures.add("topics", rel(record["path"]),
|
|
"topic %r is not one of: %s" % (topic, ", ".join(sorted(known))))
|
|
|
|
|
|
def check_numbering(failures, records):
|
|
"""The number in the filename is the number in the heading."""
|
|
for number, record in records.items():
|
|
heading = re.search(r"^# (\d+)\.", record["text"], re.M)
|
|
if not heading:
|
|
failures.add("numbering", rel(record["path"]), "no '# N. Title' heading")
|
|
elif heading.group(1) != number.lstrip("0"):
|
|
failures.add(
|
|
"numbering",
|
|
rel(record["path"]),
|
|
f"filename says {number}, heading says {heading.group(1)}",
|
|
)
|
|
|
|
|
|
def main():
|
|
failures = Failures()
|
|
records = load_records()
|
|
check_links(failures)
|
|
check_rests_on(failures, records)
|
|
check_live_citations(failures, records)
|
|
check_supersession_symmetry(failures, records)
|
|
check_numbering(failures, records)
|
|
check_topics(failures, records)
|
|
print(f"records: {len(records)} decision records checked")
|
|
return failures.report()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|