ADR 0244 sorts the mesh's concepts into ten domains (ADR 0006's contexts carry over as domains), keeps machine and node as distinct words, and makes the glossary the authority with every retired word on its replacement's Not: line. To-be 49 draws the domains; the glossary is reorganised by them, its two contradictions removed and the missing words added. words.py now fails on a retired word in running prose and on a word defined twice, so the rule is enforced rather than believed; the 63 documents it failed on are reworded here, and research 034 is kept as the record of the words it studied.
279 lines
11 KiB
Python
279 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""One word per thing, checked (ADR 0244).
|
|
|
|
The glossary (00-META/glossary.md) is the authority on the mesh's words. It names the words it
|
|
retired on one fixed kind of line, so a program can read them:
|
|
|
|
*Not:* ~~control plane~~, ~~overlay~~ (hq)
|
|
|
|
A struck word with no scope is retired everywhere; `(hq)` retires it in this repository's prose only;
|
|
`(tools)` in the descriptions of the mesh's tools only. And it names code that still carries an old
|
|
name until that code is renamed:
|
|
|
|
*Identifier until renamed:* `mesh-host` — the repository, binary and service unit
|
|
|
|
An identifier is allowed inside a code span and nowhere else.
|
|
|
|
What is enforced:
|
|
|
|
retired no retired word of scope `hq` (or none), and no identifier, in running prose of a
|
|
checked document. Running prose is what is left once code blocks, code spans, block
|
|
quotes, text in quotation marks, struck-through text, link targets, HTML comments and
|
|
frontmatter are taken out: a quotation keeps the words it quotes, a link target is a file
|
|
name, and an identifier lives in a code span.
|
|
unique no head word (a bolded word opening a glossary entry) heads two entries, and no head
|
|
word is also struck through on a *Not:* line.
|
|
tools list with `--list tools` it prints the words retired in the tools' descriptions, which is the
|
|
copy the catalogue's own check keeps (mesh-catalog `retired-words`). With
|
|
MESH_CATALOG_DIR set to a checkout of the catalogue it also fails when that copy differs.
|
|
|
|
Checked documents: 00-META/, 03-DESIGN/ (both layers), AGENTS.md and README.md, always; a research
|
|
effort initiated, or an issue opened, on or after FROM. Never 02-DECISIONS/: a decision record keeps the
|
|
words it was written with.
|
|
|
|
A document the check fails on that cannot be reworded in the change that found it is named, with a date
|
|
and a reason, in words-allowed.md beside this file. An entry past its date fails like the word itself, and
|
|
an entry for a document that no longer needs it fails too. `kept` instead of a date is allowed only for a
|
|
graduated research effort, which is a record of what was said, like a decision record.
|
|
|
|
python3 00-META/checks/words.py
|
|
python3 00-META/checks/words.py --list tools
|
|
"""
|
|
|
|
import datetime
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
ROOT = os.path.normpath(os.path.join(os.path.dirname(__file__), "..", ".."))
|
|
GLOSSARY = "00-META/glossary.md"
|
|
ALLOWED = "00-META/checks/words-allowed.md"
|
|
|
|
# The day the rule began (ADR 0244): research and issues from it on are held to it.
|
|
FROM = "2026-10-07"
|
|
|
|
ALWAYS = ("00-META/", "03-DESIGN/", "AGENTS.md", "README.md")
|
|
|
|
NOT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Not:\*(.*)$", re.M)
|
|
STRUCK = re.compile(r"~~([^~]+)~~(?:\s*\((hq|tools)\))?")
|
|
IDENT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Identifier until renamed:\*(.*)$", re.M)
|
|
CODE_SPAN = re.compile(r"`([^`]+)`")
|
|
HEAD = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*", re.M)
|
|
|
|
|
|
def rel(path):
|
|
return os.path.relpath(path, ROOT)
|
|
|
|
|
|
def read(path):
|
|
with open(os.path.join(ROOT, path), encoding="utf-8") as handle:
|
|
return handle.read()
|
|
|
|
|
|
def frontmatter_field(text, name):
|
|
if not text.startswith("---\n"):
|
|
return None
|
|
end = text.find("\n---", 4)
|
|
m = re.search(r"^%s:\s*(\S+)" % name, text[4:end], re.M)
|
|
return m.group(1) if m else None
|
|
|
|
|
|
def glossary():
|
|
"""The retired words with their scope, the identifiers, and the head words."""
|
|
text = read(GLOSSARY)
|
|
retired = []
|
|
for line in NOT_LINE.findall(text):
|
|
for word, scope in STRUCK.findall(line):
|
|
for one in word.split(" / "):
|
|
retired.append((one.strip(), scope or "all"))
|
|
identifiers = []
|
|
for line in IDENT_LINE.findall(text):
|
|
identifiers += [i.strip() for i in CODE_SPAN.findall(line)]
|
|
heads = []
|
|
for head in HEAD.findall(text):
|
|
heads += [h.strip() for h in head.split(" / ")]
|
|
return retired, identifiers, heads
|
|
|
|
|
|
def pattern(word):
|
|
"""A whole-word, case-insensitive match; a space in the word matches a space, a line break or a hyphen."""
|
|
parts = [re.escape(p) for p in word.split()]
|
|
return re.compile(r"(?i)(?<![\w-])" + r"[\s-]+".join(parts) + r"(?![\w-])")
|
|
|
|
|
|
def blank(match):
|
|
return re.sub(r"[^\n]", " ", match.group(0))
|
|
|
|
|
|
def prose(text, glossary_file=False):
|
|
"""The text with everything that is not running prose blanked, keeping offsets and line numbers."""
|
|
if text.startswith("---\n"):
|
|
end = text.find("\n---", 4)
|
|
if end != -1:
|
|
text = re.sub(r"[^\n]", " ", text[: end + 4]) + text[end + 4:]
|
|
steps = [
|
|
r"(?ms)^\s*```.*?^\s*```", # code blocks
|
|
r"(?s)<!--.*?-->", # comments
|
|
r"(?m)^\s*>.*$", # block quotes
|
|
r"`[^`\n]+`", # code spans
|
|
r"\]\([^)\s]*\)", # link targets
|
|
r"~~[^~\n]+~~", # struck through: a word named as retired
|
|
r"\"[^\"\n]*\"", # "quoted"
|
|
r"\u201c[^\u201d]*\u201d", # curly double quotes
|
|
r"\u2018[^\u2019\n]*\u2019", # curly single quotes
|
|
]
|
|
if glossary_file:
|
|
steps[:0] = [NOT_LINE.pattern, IDENT_LINE.pattern]
|
|
for step in steps:
|
|
text = re.sub(step, blank, text)
|
|
return text
|
|
|
|
|
|
def research_and_issues():
|
|
"""The research efforts and issue reports dated on or after FROM, file by file."""
|
|
out = []
|
|
for folder, key, first in (("01-RESEARCH", "initiated", "00-overview.md"), ("04-ISSUES", "opened", "00-report.md")):
|
|
base = os.path.join(ROOT, folder)
|
|
for effort in sorted(os.listdir(base)):
|
|
head = os.path.join(base, effort, first)
|
|
if not os.path.isfile(head):
|
|
continue
|
|
date = frontmatter_field(read(rel(head)), key)
|
|
if not date or date < FROM:
|
|
continue
|
|
for name in sorted(os.listdir(os.path.join(base, effort))):
|
|
if name.endswith(".md"):
|
|
out.append("%s/%s/%s" % (folder, effort, name))
|
|
return out
|
|
|
|
|
|
def checked_documents():
|
|
docs = []
|
|
for entry in ALWAYS:
|
|
path = os.path.join(ROOT, entry)
|
|
if os.path.isfile(path):
|
|
docs.append(entry)
|
|
continue
|
|
for base, dirs, files in os.walk(path):
|
|
dirs.sort()
|
|
for name in sorted(files):
|
|
if name.endswith(".md"):
|
|
docs.append(rel(os.path.join(base, name)))
|
|
return docs + research_and_issues()
|
|
|
|
|
|
def allowed():
|
|
"""words-allowed.md: one row per document — | `document` | until | why |.
|
|
|
|
`until` is a date, after which the allowance fails like the words themselves, or `kept` for a
|
|
document that records what was said and must keep its words: only a research effort that has
|
|
graduated may be kept, because once it has become a decision it is a record like one.
|
|
"""
|
|
out, problems = {}, []
|
|
if not os.path.isfile(os.path.join(ROOT, ALLOWED)):
|
|
return out, problems
|
|
today = datetime.date.today().isoformat()
|
|
for line in read(ALLOWED).splitlines():
|
|
cells = [c.strip() for c in line.strip().strip("|").split("|")]
|
|
if len(cells) != 3 or not cells[0].startswith("`"):
|
|
continue
|
|
doc, until, why = cells[0].strip("`"), cells[1], cells[2]
|
|
if not why:
|
|
problems.append("%s: the allowance for %s gives no reason" % (ALLOWED, doc))
|
|
if until == "kept":
|
|
overview = os.path.join(os.path.dirname(doc), "00-overview.md")
|
|
status = frontmatter_field(read(overview), "status") if os.path.isfile(os.path.join(ROOT, overview)) else None
|
|
if not doc.startswith("01-RESEARCH/") or status != "graduated":
|
|
problems.append("%s: %s is kept, but only a graduated research effort may be" % (ALLOWED, doc))
|
|
continue
|
|
elif not re.match(r"^\d{4}-\d{2}-\d{2}$", until):
|
|
problems.append("%s: the allowance for %s has neither a date nor `kept`" % (ALLOWED, doc))
|
|
continue
|
|
elif until < today:
|
|
problems.append("%s: the allowance for %s ran out on %s — reword it" % (ALLOWED, doc, until))
|
|
continue
|
|
out[doc] = until
|
|
return out, problems
|
|
|
|
|
|
def check_retired(retired, identifiers):
|
|
problems = []
|
|
allowance, problems_allowed = allowed()
|
|
problems += problems_allowed
|
|
words = [(w, pattern(w), "retired") for w, scope in retired if scope in ("all", "hq")]
|
|
words += [(i, pattern(i), "an identifier, allowed only in a code span") for i in identifiers]
|
|
used = set()
|
|
for doc in checked_documents():
|
|
text = prose(read(doc), glossary_file=(doc == GLOSSARY))
|
|
for word, rx, why in words:
|
|
for m in rx.finditer(text):
|
|
if doc in allowance:
|
|
used.add(doc)
|
|
continue
|
|
line = text.count("\n", 0, m.start()) + 1
|
|
problems.append("%s:%d: %r is %s (glossary)" % (doc, line, m.group(0).replace("\n", " "), why))
|
|
for doc in allowance:
|
|
if doc not in used:
|
|
problems.append("%s: %s no longer uses a retired word — remove its allowance" % (ALLOWED, doc))
|
|
return problems
|
|
|
|
|
|
def check_unique(retired, heads):
|
|
problems = []
|
|
seen = {}
|
|
for head in heads:
|
|
key = head.lower()
|
|
if key in seen:
|
|
problems.append("%s: %r heads two entries" % (GLOSSARY, head))
|
|
seen[key] = True
|
|
for word, _ in retired:
|
|
if word.lower() in seen:
|
|
problems.append("%s: %r is a head word and also retired" % (GLOSSARY, word))
|
|
return problems
|
|
|
|
|
|
def tools_list(retired):
|
|
return sorted({w for w, scope in retired if scope in ("all", "tools")}, key=str.lower)
|
|
|
|
|
|
def check_copy(retired):
|
|
catalogue = os.environ.get("MESH_CATALOG_DIR")
|
|
if not catalogue:
|
|
print("NOT COMPARED: MESH_CATALOG_DIR is not set, so the catalogue's copy of the list was not read")
|
|
return []
|
|
path = os.path.join(catalogue, "retired-words")
|
|
try:
|
|
with open(path, encoding="utf-8") as handle:
|
|
copy = [l.strip() for l in handle if l.strip() and not l.startswith("#")]
|
|
except OSError as err:
|
|
return ["the catalogue's copy of the retired words cannot be read: %s" % err]
|
|
want = tools_list(retired)
|
|
if sorted(copy, key=str.lower) != want:
|
|
return ["the catalogue's retired-words differs from the glossary: it should list %s" % ", ".join(want)]
|
|
return []
|
|
|
|
|
|
def main(argv):
|
|
retired, identifiers, heads = glossary()
|
|
if argv[1:2] == ["--list"]:
|
|
scope = argv[2] if len(argv) > 2 else "tools"
|
|
words = tools_list(retired) if scope == "tools" else sorted({w for w, s in retired if s in ("all", scope)})
|
|
print("\n".join(words))
|
|
return 0
|
|
if not retired:
|
|
print("words: the glossary names no retired word on a *Not:* line — the check would pass on anything")
|
|
return 1
|
|
problems = check_unique(retired, heads) + check_retired(retired, identifiers) + check_copy(retired)
|
|
for p in problems:
|
|
print(p)
|
|
if problems:
|
|
print("words: %d problem(s)" % len(problems))
|
|
return 1
|
|
print("words: %d retired words and %d identifiers, none in running prose; %d head words, each once"
|
|
% (len(retired), len(identifiers), len(heads)))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main(sys.argv))
|