#!/usr/bin/env python3 """Structural checks over HQ's own records. Every check here exists because the thing it checks for actually happened. See README.md for which incident is behind which check. Run from the repository root: python3 00-META/checks/records.py Exits non-zero if anything fails, so it can be a gate rather than a report. """ import os import re import sys ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # Where a citation is *guidance* rather than history. A document here tells somebody what to # do, so a link to a superseded record is an instruction to follow a withdrawn decision. # 02-DECISIONS and 01-RESEARCH are deliberately absent: they record what was decided and what # was observed, and both legitimately discuss superseded records at length. GOVERNING = ("00-META/", "03-DESIGN/01-to-be/") LINK = re.compile(r"\[[^\]]*\]\((?!https?:|mailto:)([^)]+)\)") ADR_FILE = re.compile(r"^(\d{4})-") def markdown_files(): for base, dirs, files in os.walk(ROOT): dirs[:] = [d for d in dirs if d not in (".git", ".claude")] for name in sorted(files): if name.endswith(".md"): yield os.path.join(base, name) def rel(path): return os.path.relpath(path, ROOT) def read(path): with open(path, encoding="utf-8") as handle: return handle.read() def frontmatter(text): """Minimal frontmatter reader — enough for the fields these checks use. Not a YAML parser on purpose: a dependency in a repository that has none, to read four scalar fields and one list, would cost more than it returns. """ if not text.startswith("---\n"): return {} end = text.find("\n---", 4) if end == -1: return {} fields, key = {}, None for line in text[4:end].split("\n"): item = re.match(r"^\s+-\s+(.*)$", line) if item and key: fields.setdefault(key, []).append(item.group(1).strip()) continue pair = re.match(r"^([A-Za-z_-]+):\s*(.*)$", line) if pair: key = pair.group(1) value = pair.group(2).strip() if value.startswith("[") and value.endswith("]"): # inline list: `decisions: [a, b]`, `code: []` inner = value[1:-1].strip() fields[key] = [v.strip() for v in inner.split(",") if v.strip()] elif value: fields[key] = value else: fields[key] = [] return fields def load_records(): """Every decision record, by its four-digit number.""" records = {} folder = os.path.join(ROOT, "02-DECISIONS") for name in sorted(os.listdir(folder)): match = ADR_FILE.match(name) if not name.endswith(".md") or not match: continue path = os.path.join(folder, name) text = read(path) records[match.group(1)] = { "number": match.group(1), "name": name, "path": path, "text": text, "front": frontmatter(text), } return records class Failures: def __init__(self): self.items = [] def add(self, check, location, message): self.items.append((check, location, message)) def report(self): if not self.items: print("records: all checks passed") return 0 by_check = {} for check, location, message in self.items: by_check.setdefault(check, []).append((location, message)) for check in sorted(by_check): print(f"\n{check} — {len(by_check[check])} problem(s)") for location, message in by_check[check]: print(f" {location}\n {message}") print(f"\nrecords: {len(self.items)} problem(s)") return 1 def check_links(failures): """Every relative link resolves to something that exists.""" for path in markdown_files(): folder = os.path.dirname(path) for number, line in enumerate(read(path).split("\n"), 1): for target in LINK.findall(line): target = target.split("#")[0].strip() if not target: continue if not os.path.exists(os.path.normpath(os.path.join(folder, target))): failures.add("links", f"{rel(path)}:{number}", f"link does not resolve: {target}") def check_rests_on(failures, records): """A document's `decisions:` and a record's `extends:` must name a live record. These are the load-bearing citations: the document declares that it rests on that decision. Resting on a withdrawn one is the defect this whole check set exists for. """ for path in markdown_files(): front = frontmatter(read(path)) cited = list(front.get("decisions", []) or []) extends = front.get("extends") if isinstance(extends, str) and extends: cited.append(extends) for entry in cited: match = ADR_FILE.match(os.path.basename(entry)) if not match: failures.add("rests-on", rel(path), f"not a decision record: {entry}") continue number = match.group(1) if number not in records: failures.add("rests-on", rel(path), f"no such record: {entry}") continue status = records[number]["front"].get("status") if status != "accepted": # A proposed record may extend another proposed one. Decisions are drafted in # chains -- 0059 extends 0057 while both await review -- and refusing that would # mean either drafting out of order or marking records accepted to satisfy a # check, which is the failure this repository already made once. if frontmatter(read(path)).get("status") == "proposed": continue # An as-is document describes what runs, and what runs was built under # whatever was decided at the time. ADR 0056: "as-is describing a superseded # decision is exactly what as-is is for." if rel(path).startswith("03-DESIGN/00-as-is/"): continue # An extension that supersedes legitimately names what it replaced. this = ADR_FILE.match(os.path.basename(path)) supersedes = records[number]["front"].get("superseded-by", "") if this and supersedes and os.path.basename(path) in str(supersedes): continue failures.add( "rests-on", rel(path), f"rests on ADR {number}, which is '{status}' — a document may not rest on a " f"record that is not accepted", ) def check_live_citations(failures, records): """In a governing document, a link to a superseded record must name its replacement. The reader of a rule needs to know the rule was withdrawn, and needs somewhere to go. Naming the superseder in the same paragraph is both, and it is what a person would write anyway. """ superseded = { number: record["front"].get("superseded-by", "") for number, record in records.items() if record["front"].get("status") == "superseded" } for path in markdown_files(): if not any(rel(path).startswith(prefix) for prefix in GOVERNING): continue text = read(path) offset = 0 for paragraph in text.split("\n\n"): line_no = text[:offset].count("\n") + 1 offset += len(paragraph) + 2 targets = LINK.findall(paragraph) for target in targets: match = ADR_FILE.match(os.path.basename(target.split("#")[0])) if not match or match.group(1) not in superseded: continue number = match.group(1) replacement = os.path.basename(str(superseded[number])) if not replacement: failures.add( "live-citation", f"{rel(path)}:{line_no}", f"cites superseded ADR {number}, which names no superseder", ) continue if not any(replacement in t for t in targets): failures.add( "live-citation", f"{rel(path)}:{line_no}", f"cites superseded ADR {number} without naming its replacement " f"({replacement}) in the same paragraph", ) def check_supersession_symmetry(failures, records): """If A says it was superseded by B, B must say it supersedes A.""" for number, record in records.items(): front = record["front"] status = front.get("status") by = os.path.basename(str(front.get("superseded-by", ""))) if status == "superseded" and not by: failures.add("supersession", rel(record["path"]), "marked superseded but names no superseder") continue if by and status != "superseded": failures.add("supersession", rel(record["path"]), f"names a superseder but status is '{status}'") if not by: continue match = ADR_FILE.match(by) if not match or match.group(1) not in records: failures.add("supersession", rel(record["path"]), f"superseder does not exist: {by}") continue other = records[match.group(1)] claims = os.path.basename(str(other["front"].get("supersedes", ""))) if claims != record["name"]: failures.add( "supersession", rel(other["path"]), f"ADR {number} says this supersedes it; this record does not say so " f"(supersedes: {claims or 'absent'})", ) def check_topics(failures, records): """Every record names a topic the index knows. The topic is what puts a record in the reading order, so a record without one — or with one nobody defined — disappears from the index rather than appearing in the wrong place. That is the quiet failure, so it is the one checked. """ known = {"the mesh", "the tiers", "what runs on it", "building it", "checking it", "how we work"} for number, record in sorted(records.items()): topic = record["front"].get("topic") if not topic: failures.add("topics", rel(record["path"]), "no topic, so it has no place in the reading order") elif topic not in known: failures.add("topics", rel(record["path"]), "topic %r is not one of: %s" % (topic, ", ".join(sorted(known)))) def check_numbering(failures, records): """The number in the filename is the number in the heading.""" for number, record in records.items(): heading = re.search(r"^# (\d+)\.", record["text"], re.M) if not heading: failures.add("numbering", rel(record["path"]), "no '# N. Title' heading") elif heading.group(1) != number.lstrip("0"): failures.add( "numbering", rel(record["path"]), f"filename says {number}, heading says {heading.group(1)}", ) def check_progressive_insights(failures, records): """A correction made inside a record is marked and dated, or it is a silent rewrite. A record may be corrected in place when a *fact* in it went stale and the decision still stands (`02-DECISIONS/README.md`, "Progressive insight"). The whole safety of that allowance is that the correction is legible in the record rather than only in a diff nobody reads, so the form is what is checked here: every mention of an insight is the marker, the marker carries an ISO date, and that date is not earlier than the decision's own — an insight predating the decision it corrects is a copied marker, not a correction. What this cannot check is an edit made with no marker at all. Nothing mechanical can; that one is the reviewer's, reading the diff. The check keeps the *marked* path honest so that an unmarked change stands out as the anomaly it is. """ phrase = re.compile(r"progressive insight", re.I) # Both patterns stay on one line: a bold run does not span paragraphs, and `[^*]*` across # newlines will happily join an unrelated `**` far above to the marker below, reporting the # whole span between them. It did exactly that the first time this ran. marker = re.compile(r"\*\*Progressive insights?[ \t]*[\u2014\u2013-][ \t]*(\d{4}-\d{2}-\d{2})\.?\*\*") loose = re.compile(r"\*\*[^*\n]*[Pp]rogressive insights?[^*\n]*\*\*") iso = re.compile(r"^\d{4}-\d{2}-\d{2}$") for number, record in sorted(records.items()): text = record["text"] if not phrase.search(text): continue decided = str(record["front"].get("date", "")) good = [(m.start(), m.end(), m.group(1)) for m in marker.finditer(text)] for m in loose.finditer(text): if any(s <= m.start() and m.end() <= e for s, e, _ in good): continue failures.add("insights", rel(record["path"]), "a progressive insight is not in the dated marked form " "'**Progressive insight \u2014 YYYY-MM-DD.**': %s" % m.group(0)) for _, _, stamp in good: if decided and iso.match(decided) and stamp < decided: failures.add("insights", rel(record["path"]), "a progressive insight dated %s predates the decision (%s)" % (stamp, decided)) covered = [(s, e) for s, e, _ in good] for m in phrase.finditer(text): if any(s <= m.start() and m.end() <= e for s, e in covered): continue line = text.rfind("\n", 0, m.start()) + 1 end = text.find("\n", m.end()) whole = text[line:end if end != -1 else len(text)] if whole.lstrip().startswith("#"): continue # A line that also carries a link is discussing the rule, not marking a correction: # a marker never needs to cite anything, and a record that reasons about the policy # must be able to name it. Bare prose with no citation is the informal marking this # is here to catch. if "](" in whole: continue if loose.search(text, line, text.find("\n", m.end()) + 1 or len(text)): continue failures.add("insights", rel(record["path"]), "'progressive insight' appears unmarked; a correction is marked and " "dated, or it is a silent rewrite") def check_status_against_code(failures): """A design document naming specific code may not still call itself `designed`. **Naming a file is a claim that the file implements this**, so the two fields have to agree. They drifted: ten to-be documents named working, lab-proven code — several with a *What was built* or *Raised, and observed* section — while still saying nothing had been built. Deliberately weak, and that is the point of it being mechanical. It cannot tell whether the prose is true, only that a document has stopped claiming to be unbuilt once it points at something. `code: [mesh-controller]` — a repository with no path — is a plan and stays `designed`. """ for path in markdown_files(): if not rel(path).startswith("03-DESIGN/01-to-be/") or path.endswith("README.md"): continue front = frontmatter(read(path)) if front.get("status") != "designed": continue for entry in front.get("code") or []: named = re.sub(r"\s*\(.*\)$", "", entry).strip().split(None, 1) if len(named) > 1: failures.add( "status-vs-code", rel(path), f"`designed`, but names {named[1]!r} in {named[0]}. Naming a file claims " f"it implements this — use `in-progress`, or `implemented` once it is " f"defensible from that repository's main branch.", ) break def main(): failures = Failures() records = load_records() check_links(failures) check_rests_on(failures, records) check_live_citations(failures, records) check_supersession_symmetry(failures, records) check_numbering(failures, records) check_topics(failures, records) check_status_against_code(failures) check_progressive_insights(failures, records) print(f"records: {len(records)} decision records checked") return failures.report() if __name__ == "__main__": sys.exit(main())