Files
hq/00-META/checks/words.py
T
jochen 350f7dcdc2 Hold a retired word's plural to the word, so alerts and rollouts no longer pass
words.py matched a retired word only when nothing followed it, so the
plural of every retired word passed. The pattern now takes s or es on the
last word; 14 uses in 9 documents are reworded, two of them 'the hosts'
for the node-engines.
2026-10-07 21:14:36 +02:00

284 lines
12 KiB
Python

#!/usr/bin/env python3
"""One word per thing, checked (ADR 0244).
The glossary (00-META/glossary.md) is the authority on the mesh's words. It names the words it
retired on one fixed kind of line, so a program can read them:
*Not:* ~~control plane~~, ~~overlay~~ (hq)
A struck word with no scope is retired everywhere; `(hq)` retires it in this repository's prose only;
`(tools)` in the descriptions of the mesh's tools only. And it names code that still carries an old
name until that code is renamed:
*Identifier until renamed:* `mesh-host` — the repository, binary and service unit
An identifier is allowed inside a code span and nowhere else.
What is enforced:
retired no retired word of scope `hq` (or none), and no identifier, in running prose of a
checked document. Running prose is what is left once code blocks, code spans, block
quotes, text in quotation marks, struck-through text, link targets, HTML comments and
frontmatter are taken out: a quotation keeps the words it quotes, a link target is a file
name, and an identifier lives in a code span. A word's plural is the word: "alerts" fails
as "alert" does.
unique no head word (a bolded word opening a glossary entry) heads two entries, and no head
word is also struck through on a *Not:* line.
tools list with `--list tools` it prints the words retired in the tools' descriptions, which is the
copy the catalogue's own check keeps (mesh-catalog `retired-words`). With
MESH_CATALOG_DIR set to a checkout of the catalogue it also fails when that copy differs.
Checked documents: 00-META/, 03-DESIGN/ (both layers), AGENTS.md and README.md, always; a research
effort initiated, or an issue opened, on or after FROM. Never 02-DECISIONS/: a decision record keeps the
words it was written with.
A document the check fails on that cannot be reworded in the change that found it is named, with a date
and a reason, in words-allowed.md beside this file. An entry past its date fails like the word itself, and
an entry for a document that no longer needs it fails too. `kept` instead of a date is allowed only for a
graduated research effort, which is a record of what was said, like a decision record.
python3 00-META/checks/words.py
python3 00-META/checks/words.py --list tools
"""
import datetime
import os
import re
import sys
ROOT = os.path.normpath(os.path.join(os.path.dirname(__file__), "..", ".."))
GLOSSARY = "00-META/glossary.md"
ALLOWED = "00-META/checks/words-allowed.md"
# The day the rule began (ADR 0244): research and issues from it on are held to it.
FROM = "2026-10-07"
ALWAYS = ("00-META/", "03-DESIGN/", "AGENTS.md", "README.md")
NOT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Not:\*(.*)$", re.M)
STRUCK = re.compile(r"~~([^~]+)~~(?:\s*\((hq|tools)\))?")
IDENT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Identifier until renamed:\*(.*)$", re.M)
CODE_SPAN = re.compile(r"`([^`]+)`")
HEAD = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*", re.M)
def rel(path):
return os.path.relpath(path, ROOT)
def read(path):
with open(os.path.join(ROOT, path), encoding="utf-8") as handle:
return handle.read()
def frontmatter_field(text, name):
if not text.startswith("---\n"):
return None
end = text.find("\n---", 4)
m = re.search(r"^%s:\s*(\S+)" % name, text[4:end], re.M)
return m.group(1) if m else None
def glossary():
"""The retired words with their scope, the identifiers, and the head words."""
text = read(GLOSSARY)
retired = []
for line in NOT_LINE.findall(text):
for word, scope in STRUCK.findall(line):
for one in word.split(" / "):
retired.append((one.strip(), scope or "all"))
identifiers = []
for line in IDENT_LINE.findall(text):
identifiers += [i.strip() for i in CODE_SPAN.findall(line)]
heads = []
for head in HEAD.findall(text):
heads += [h.strip() for h in head.split(" / ")]
return retired, identifiers, heads
def pattern(word):
"""A whole-word, case-insensitive match; a space in the word matches a space, a line break or a hyphen.
The plural is the word too: its last part may end in `s` or `es` ("alerts", "control planes"), since
a retired word does not come back by being counted. Anything else joined on is another word.
"""
parts = [re.escape(p) for p in word.split()]
return re.compile(r"(?i)(?<![\w-])" + r"[\s-]+".join(parts) + r"(?:e?s)?(?![\w-])")
def blank(match):
return re.sub(r"[^\n]", " ", match.group(0))
def prose(text, glossary_file=False):
"""The text with everything that is not running prose blanked, keeping offsets and line numbers."""
if text.startswith("---\n"):
end = text.find("\n---", 4)
if end != -1:
text = re.sub(r"[^\n]", " ", text[: end + 4]) + text[end + 4:]
steps = [
r"(?ms)^\s*```.*?^\s*```", # code blocks
r"(?s)<!--.*?-->", # comments
r"(?m)^\s*>.*$", # block quotes
r"`[^`\n]+`", # code spans
r"\]\([^)\s]*\)", # link targets
r"~~[^~\n]+~~", # struck through: a word named as retired
r"\"[^\"\n]*\"", # "quoted"
r"\u201c[^\u201d]*\u201d", # curly double quotes
r"\u2018[^\u2019\n]*\u2019", # curly single quotes
]
if glossary_file:
steps[:0] = [NOT_LINE.pattern, IDENT_LINE.pattern]
for step in steps:
text = re.sub(step, blank, text)
return text
def research_and_issues():
"""The research efforts and issue reports dated on or after FROM, file by file."""
out = []
for folder, key, first in (("01-RESEARCH", "initiated", "00-overview.md"), ("04-ISSUES", "opened", "00-report.md")):
base = os.path.join(ROOT, folder)
for effort in sorted(os.listdir(base)):
head = os.path.join(base, effort, first)
if not os.path.isfile(head):
continue
date = frontmatter_field(read(rel(head)), key)
if not date or date < FROM:
continue
for name in sorted(os.listdir(os.path.join(base, effort))):
if name.endswith(".md"):
out.append("%s/%s/%s" % (folder, effort, name))
return out
def checked_documents():
docs = []
for entry in ALWAYS:
path = os.path.join(ROOT, entry)
if os.path.isfile(path):
docs.append(entry)
continue
for base, dirs, files in os.walk(path):
dirs.sort()
for name in sorted(files):
if name.endswith(".md"):
docs.append(rel(os.path.join(base, name)))
return docs + research_and_issues()
def allowed():
"""words-allowed.md: one row per document — | `document` | until | why |.
`until` is a date, after which the allowance fails like the words themselves, or `kept` for a
document that records what was said and must keep its words: only a research effort that has
graduated may be kept, because once it has become a decision it is a record like one.
"""
out, problems = {}, []
if not os.path.isfile(os.path.join(ROOT, ALLOWED)):
return out, problems
today = datetime.date.today().isoformat()
for line in read(ALLOWED).splitlines():
cells = [c.strip() for c in line.strip().strip("|").split("|")]
if len(cells) != 3 or not cells[0].startswith("`"):
continue
doc, until, why = cells[0].strip("`"), cells[1], cells[2]
if not why:
problems.append("%s: the allowance for %s gives no reason" % (ALLOWED, doc))
if until == "kept":
overview = os.path.join(os.path.dirname(doc), "00-overview.md")
status = frontmatter_field(read(overview), "status") if os.path.isfile(os.path.join(ROOT, overview)) else None
if not doc.startswith("01-RESEARCH/") or status != "graduated":
problems.append("%s: %s is kept, but only a graduated research effort may be" % (ALLOWED, doc))
continue
elif not re.match(r"^\d{4}-\d{2}-\d{2}$", until):
problems.append("%s: the allowance for %s has neither a date nor `kept`" % (ALLOWED, doc))
continue
elif until < today:
problems.append("%s: the allowance for %s ran out on %s — reword it" % (ALLOWED, doc, until))
continue
out[doc] = until
return out, problems
def check_retired(retired, identifiers):
problems = []
allowance, problems_allowed = allowed()
problems += problems_allowed
words = [(w, pattern(w), "retired") for w, scope in retired if scope in ("all", "hq")]
words += [(i, pattern(i), "an identifier, allowed only in a code span") for i in identifiers]
used = set()
for doc in checked_documents():
text = prose(read(doc), glossary_file=(doc == GLOSSARY))
for word, rx, why in words:
for m in rx.finditer(text):
if doc in allowance:
used.add(doc)
continue
line = text.count("\n", 0, m.start()) + 1
problems.append("%s:%d: %r is %s (glossary)" % (doc, line, m.group(0).replace("\n", " "), why))
for doc in allowance:
if doc not in used:
problems.append("%s: %s no longer uses a retired word — remove its allowance" % (ALLOWED, doc))
return problems
def check_unique(retired, heads):
problems = []
seen = {}
for head in heads:
key = head.lower()
if key in seen:
problems.append("%s: %r heads two entries" % (GLOSSARY, head))
seen[key] = True
for word, _ in retired:
if word.lower() in seen:
problems.append("%s: %r is a head word and also retired" % (GLOSSARY, word))
return problems
def tools_list(retired):
return sorted({w for w, scope in retired if scope in ("all", "tools")}, key=str.lower)
def check_copy(retired):
catalogue = os.environ.get("MESH_CATALOG_DIR")
if not catalogue:
print("NOT COMPARED: MESH_CATALOG_DIR is not set, so the catalogue's copy of the list was not read")
return []
path = os.path.join(catalogue, "retired-words")
try:
with open(path, encoding="utf-8") as handle:
copy = [l.strip() for l in handle if l.strip() and not l.startswith("#")]
except OSError as err:
return ["the catalogue's copy of the retired words cannot be read: %s" % err]
want = tools_list(retired)
if sorted(copy, key=str.lower) != want:
return ["the catalogue's retired-words differs from the glossary: it should list %s" % ", ".join(want)]
return []
def main(argv):
retired, identifiers, heads = glossary()
if argv[1:2] == ["--list"]:
scope = argv[2] if len(argv) > 2 else "tools"
words = tools_list(retired) if scope == "tools" else sorted({w for w, s in retired if s in ("all", scope)})
print("\n".join(words))
return 0
if not retired:
print("words: the glossary names no retired word on a *Not:* line — the check would pass on anything")
return 1
problems = check_unique(retired, heads) + check_retired(retired, identifiers) + check_copy(retired)
for p in problems:
print(p)
if problems:
print("words: %d problem(s)" % len(problems))
return 1
print("words: %d retired words and %d identifiers, none in running prose; %d head words, each once"
% (len(retired), len(identifiers), len(heads)))
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv))