Files
hq/00-META/checks/words.py
T
jochen a28da1734b Graduate research 034: the mesh in domains, one word per thing, checked
ADR 0244 sorts the mesh's concepts into ten domains (ADR 0006's contexts
carry over as domains), keeps machine and node as distinct words, and
makes the glossary the authority with every retired word on its
replacement's Not: line. To-be 49 draws the domains; the glossary is
reorganised by them, its two contradictions removed and the missing
words added. words.py now fails on a retired word in running prose and
on a word defined twice, so the rule is enforced rather than believed;
the 63 documents it failed on are reworded here, and research 034 is
kept as the record of the words it studied.
2026-10-07 19:58:39 +02:00

279 lines
11 KiB
Python

#!/usr/bin/env python3
"""One word per thing, checked (ADR 0244).
The glossary (00-META/glossary.md) is the authority on the mesh's words. It names the words it
retired on one fixed kind of line, so a program can read them:
*Not:* ~~control plane~~, ~~overlay~~ (hq)
A struck word with no scope is retired everywhere; `(hq)` retires it in this repository's prose only;
`(tools)` in the descriptions of the mesh's tools only. And it names code that still carries an old
name until that code is renamed:
*Identifier until renamed:* `mesh-host` — the repository, binary and service unit
An identifier is allowed inside a code span and nowhere else.
What is enforced:
retired no retired word of scope `hq` (or none), and no identifier, in running prose of a
checked document. Running prose is what is left once code blocks, code spans, block
quotes, text in quotation marks, struck-through text, link targets, HTML comments and
frontmatter are taken out: a quotation keeps the words it quotes, a link target is a file
name, and an identifier lives in a code span.
unique no head word (a bolded word opening a glossary entry) heads two entries, and no head
word is also struck through on a *Not:* line.
tools list with `--list tools` it prints the words retired in the tools' descriptions, which is the
copy the catalogue's own check keeps (mesh-catalog `retired-words`). With
MESH_CATALOG_DIR set to a checkout of the catalogue it also fails when that copy differs.
Checked documents: 00-META/, 03-DESIGN/ (both layers), AGENTS.md and README.md, always; a research
effort initiated, or an issue opened, on or after FROM. Never 02-DECISIONS/: a decision record keeps the
words it was written with.
A document the check fails on that cannot be reworded in the change that found it is named, with a date
and a reason, in words-allowed.md beside this file. An entry past its date fails like the word itself, and
an entry for a document that no longer needs it fails too. `kept` instead of a date is allowed only for a
graduated research effort, which is a record of what was said, like a decision record.
python3 00-META/checks/words.py
python3 00-META/checks/words.py --list tools
"""
import datetime
import os
import re
import sys
ROOT = os.path.normpath(os.path.join(os.path.dirname(__file__), "..", ".."))
GLOSSARY = "00-META/glossary.md"
ALLOWED = "00-META/checks/words-allowed.md"
# The day the rule began (ADR 0244): research and issues from it on are held to it.
FROM = "2026-10-07"
ALWAYS = ("00-META/", "03-DESIGN/", "AGENTS.md", "README.md")
NOT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Not:\*(.*)$", re.M)
STRUCK = re.compile(r"~~([^~]+)~~(?:\s*\((hq|tools)\))?")
IDENT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Identifier until renamed:\*(.*)$", re.M)
CODE_SPAN = re.compile(r"`([^`]+)`")
HEAD = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*", re.M)
def rel(path):
return os.path.relpath(path, ROOT)
def read(path):
with open(os.path.join(ROOT, path), encoding="utf-8") as handle:
return handle.read()
def frontmatter_field(text, name):
if not text.startswith("---\n"):
return None
end = text.find("\n---", 4)
m = re.search(r"^%s:\s*(\S+)" % name, text[4:end], re.M)
return m.group(1) if m else None
def glossary():
"""The retired words with their scope, the identifiers, and the head words."""
text = read(GLOSSARY)
retired = []
for line in NOT_LINE.findall(text):
for word, scope in STRUCK.findall(line):
for one in word.split(" / "):
retired.append((one.strip(), scope or "all"))
identifiers = []
for line in IDENT_LINE.findall(text):
identifiers += [i.strip() for i in CODE_SPAN.findall(line)]
heads = []
for head in HEAD.findall(text):
heads += [h.strip() for h in head.split(" / ")]
return retired, identifiers, heads
def pattern(word):
"""A whole-word, case-insensitive match; a space in the word matches a space, a line break or a hyphen."""
parts = [re.escape(p) for p in word.split()]
return re.compile(r"(?i)(?<![\w-])" + r"[\s-]+".join(parts) + r"(?![\w-])")
def blank(match):
return re.sub(r"[^\n]", " ", match.group(0))
def prose(text, glossary_file=False):
"""The text with everything that is not running prose blanked, keeping offsets and line numbers."""
if text.startswith("---\n"):
end = text.find("\n---", 4)
if end != -1:
text = re.sub(r"[^\n]", " ", text[: end + 4]) + text[end + 4:]
steps = [
r"(?ms)^\s*```.*?^\s*```", # code blocks
r"(?s)<!--.*?-->", # comments
r"(?m)^\s*>.*$", # block quotes
r"`[^`\n]+`", # code spans
r"\]\([^)\s]*\)", # link targets
r"~~[^~\n]+~~", # struck through: a word named as retired
r"\"[^\"\n]*\"", # "quoted"
r"\u201c[^\u201d]*\u201d", # curly double quotes
r"\u2018[^\u2019\n]*\u2019", # curly single quotes
]
if glossary_file:
steps[:0] = [NOT_LINE.pattern, IDENT_LINE.pattern]
for step in steps:
text = re.sub(step, blank, text)
return text
def research_and_issues():
"""The research efforts and issue reports dated on or after FROM, file by file."""
out = []
for folder, key, first in (("01-RESEARCH", "initiated", "00-overview.md"), ("04-ISSUES", "opened", "00-report.md")):
base = os.path.join(ROOT, folder)
for effort in sorted(os.listdir(base)):
head = os.path.join(base, effort, first)
if not os.path.isfile(head):
continue
date = frontmatter_field(read(rel(head)), key)
if not date or date < FROM:
continue
for name in sorted(os.listdir(os.path.join(base, effort))):
if name.endswith(".md"):
out.append("%s/%s/%s" % (folder, effort, name))
return out
def checked_documents():
docs = []
for entry in ALWAYS:
path = os.path.join(ROOT, entry)
if os.path.isfile(path):
docs.append(entry)
continue
for base, dirs, files in os.walk(path):
dirs.sort()
for name in sorted(files):
if name.endswith(".md"):
docs.append(rel(os.path.join(base, name)))
return docs + research_and_issues()
def allowed():
"""words-allowed.md: one row per document — | `document` | until | why |.
`until` is a date, after which the allowance fails like the words themselves, or `kept` for a
document that records what was said and must keep its words: only a research effort that has
graduated may be kept, because once it has become a decision it is a record like one.
"""
out, problems = {}, []
if not os.path.isfile(os.path.join(ROOT, ALLOWED)):
return out, problems
today = datetime.date.today().isoformat()
for line in read(ALLOWED).splitlines():
cells = [c.strip() for c in line.strip().strip("|").split("|")]
if len(cells) != 3 or not cells[0].startswith("`"):
continue
doc, until, why = cells[0].strip("`"), cells[1], cells[2]
if not why:
problems.append("%s: the allowance for %s gives no reason" % (ALLOWED, doc))
if until == "kept":
overview = os.path.join(os.path.dirname(doc), "00-overview.md")
status = frontmatter_field(read(overview), "status") if os.path.isfile(os.path.join(ROOT, overview)) else None
if not doc.startswith("01-RESEARCH/") or status != "graduated":
problems.append("%s: %s is kept, but only a graduated research effort may be" % (ALLOWED, doc))
continue
elif not re.match(r"^\d{4}-\d{2}-\d{2}$", until):
problems.append("%s: the allowance for %s has neither a date nor `kept`" % (ALLOWED, doc))
continue
elif until < today:
problems.append("%s: the allowance for %s ran out on %s — reword it" % (ALLOWED, doc, until))
continue
out[doc] = until
return out, problems
def check_retired(retired, identifiers):
problems = []
allowance, problems_allowed = allowed()
problems += problems_allowed
words = [(w, pattern(w), "retired") for w, scope in retired if scope in ("all", "hq")]
words += [(i, pattern(i), "an identifier, allowed only in a code span") for i in identifiers]
used = set()
for doc in checked_documents():
text = prose(read(doc), glossary_file=(doc == GLOSSARY))
for word, rx, why in words:
for m in rx.finditer(text):
if doc in allowance:
used.add(doc)
continue
line = text.count("\n", 0, m.start()) + 1
problems.append("%s:%d: %r is %s (glossary)" % (doc, line, m.group(0).replace("\n", " "), why))
for doc in allowance:
if doc not in used:
problems.append("%s: %s no longer uses a retired word — remove its allowance" % (ALLOWED, doc))
return problems
def check_unique(retired, heads):
problems = []
seen = {}
for head in heads:
key = head.lower()
if key in seen:
problems.append("%s: %r heads two entries" % (GLOSSARY, head))
seen[key] = True
for word, _ in retired:
if word.lower() in seen:
problems.append("%s: %r is a head word and also retired" % (GLOSSARY, word))
return problems
def tools_list(retired):
return sorted({w for w, scope in retired if scope in ("all", "tools")}, key=str.lower)
def check_copy(retired):
catalogue = os.environ.get("MESH_CATALOG_DIR")
if not catalogue:
print("NOT COMPARED: MESH_CATALOG_DIR is not set, so the catalogue's copy of the list was not read")
return []
path = os.path.join(catalogue, "retired-words")
try:
with open(path, encoding="utf-8") as handle:
copy = [l.strip() for l in handle if l.strip() and not l.startswith("#")]
except OSError as err:
return ["the catalogue's copy of the retired words cannot be read: %s" % err]
want = tools_list(retired)
if sorted(copy, key=str.lower) != want:
return ["the catalogue's retired-words differs from the glossary: it should list %s" % ", ".join(want)]
return []
def main(argv):
retired, identifiers, heads = glossary()
if argv[1:2] == ["--list"]:
scope = argv[2] if len(argv) > 2 else "tools"
words = tools_list(retired) if scope == "tools" else sorted({w for w, s in retired if s in ("all", scope)})
print("\n".join(words))
return 0
if not retired:
print("words: the glossary names no retired word on a *Not:* line — the check would pass on anything")
return 1
problems = check_unique(retired, heads) + check_retired(retired, identifiers) + check_copy(retired)
for p in problems:
print(p)
if problems:
print("words: %d problem(s)" % len(problems))
return 1
print("words: %d retired words and %d identifiers, none in running prose; %d head words, each once"
% (len(retired), len(identifiers), len(heads)))
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv))