words.py matched a retired word only when nothing followed it, so the plural of every retired word passed. The pattern now takes s or es on the last word; 14 uses in 9 documents are reworded, two of them 'the hosts' for the node-engines.
284 lines
12 KiB
Python
284 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""One word per thing, checked (ADR 0244).
|
|
|
|
The glossary (00-META/glossary.md) is the authority on the mesh's words. It names the words it
|
|
retired on one fixed kind of line, so a program can read them:
|
|
|
|
*Not:* ~~control plane~~, ~~overlay~~ (hq)
|
|
|
|
A struck word with no scope is retired everywhere; `(hq)` retires it in this repository's prose only;
|
|
`(tools)` in the descriptions of the mesh's tools only. And it names code that still carries an old
|
|
name until that code is renamed:
|
|
|
|
*Identifier until renamed:* `mesh-host` — the repository, binary and service unit
|
|
|
|
An identifier is allowed inside a code span and nowhere else.
|
|
|
|
What is enforced:
|
|
|
|
retired no retired word of scope `hq` (or none), and no identifier, in running prose of a
|
|
checked document. Running prose is what is left once code blocks, code spans, block
|
|
quotes, text in quotation marks, struck-through text, link targets, HTML comments and
|
|
frontmatter are taken out: a quotation keeps the words it quotes, a link target is a file
|
|
name, and an identifier lives in a code span. A word's plural is the word: "alerts" fails
|
|
as "alert" does.
|
|
unique no head word (a bolded word opening a glossary entry) heads two entries, and no head
|
|
word is also struck through on a *Not:* line.
|
|
tools list with `--list tools` it prints the words retired in the tools' descriptions, which is the
|
|
copy the catalogue's own check keeps (mesh-catalog `retired-words`). With
|
|
MESH_CATALOG_DIR set to a checkout of the catalogue it also fails when that copy differs.
|
|
|
|
Checked documents: 00-META/, 03-DESIGN/ (both layers), AGENTS.md and README.md, always; a research
|
|
effort initiated, or an issue opened, on or after FROM. Never 02-DECISIONS/: a decision record keeps the
|
|
words it was written with.
|
|
|
|
A document the check fails on that cannot be reworded in the change that found it is named, with a date
|
|
and a reason, in words-allowed.md beside this file. An entry past its date fails like the word itself, and
|
|
an entry for a document that no longer needs it fails too. `kept` instead of a date is allowed only for a
|
|
graduated research effort, which is a record of what was said, like a decision record.
|
|
|
|
python3 00-META/checks/words.py
|
|
python3 00-META/checks/words.py --list tools
|
|
"""
|
|
|
|
import datetime
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
ROOT = os.path.normpath(os.path.join(os.path.dirname(__file__), "..", ".."))
|
|
GLOSSARY = "00-META/glossary.md"
|
|
ALLOWED = "00-META/checks/words-allowed.md"
|
|
|
|
# The day the rule began (ADR 0244): research and issues from it on are held to it.
|
|
FROM = "2026-10-07"
|
|
|
|
ALWAYS = ("00-META/", "03-DESIGN/", "AGENTS.md", "README.md")
|
|
|
|
NOT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Not:\*(.*)$", re.M)
|
|
STRUCK = re.compile(r"~~([^~]+)~~(?:\s*\((hq|tools)\))?")
|
|
IDENT_LINE = re.compile(r"^\s*(?:[-*]\s+)?\*Identifier until renamed:\*(.*)$", re.M)
|
|
CODE_SPAN = re.compile(r"`([^`]+)`")
|
|
HEAD = re.compile(r"^\s*-\s+\*\*([^*]+)\*\*", re.M)
|
|
|
|
|
|
def rel(path):
|
|
return os.path.relpath(path, ROOT)
|
|
|
|
|
|
def read(path):
|
|
with open(os.path.join(ROOT, path), encoding="utf-8") as handle:
|
|
return handle.read()
|
|
|
|
|
|
def frontmatter_field(text, name):
|
|
if not text.startswith("---\n"):
|
|
return None
|
|
end = text.find("\n---", 4)
|
|
m = re.search(r"^%s:\s*(\S+)" % name, text[4:end], re.M)
|
|
return m.group(1) if m else None
|
|
|
|
|
|
def glossary():
|
|
"""The retired words with their scope, the identifiers, and the head words."""
|
|
text = read(GLOSSARY)
|
|
retired = []
|
|
for line in NOT_LINE.findall(text):
|
|
for word, scope in STRUCK.findall(line):
|
|
for one in word.split(" / "):
|
|
retired.append((one.strip(), scope or "all"))
|
|
identifiers = []
|
|
for line in IDENT_LINE.findall(text):
|
|
identifiers += [i.strip() for i in CODE_SPAN.findall(line)]
|
|
heads = []
|
|
for head in HEAD.findall(text):
|
|
heads += [h.strip() for h in head.split(" / ")]
|
|
return retired, identifiers, heads
|
|
|
|
|
|
def pattern(word):
|
|
"""A whole-word, case-insensitive match; a space in the word matches a space, a line break or a hyphen.
|
|
|
|
The plural is the word too: its last part may end in `s` or `es` ("alerts", "control planes"), since
|
|
a retired word does not come back by being counted. Anything else joined on is another word.
|
|
"""
|
|
parts = [re.escape(p) for p in word.split()]
|
|
return re.compile(r"(?i)(?<![\w-])" + r"[\s-]+".join(parts) + r"(?:e?s)?(?![\w-])")
|
|
|
|
|
|
def blank(match):
|
|
return re.sub(r"[^\n]", " ", match.group(0))
|
|
|
|
|
|
def prose(text, glossary_file=False):
|
|
"""The text with everything that is not running prose blanked, keeping offsets and line numbers."""
|
|
if text.startswith("---\n"):
|
|
end = text.find("\n---", 4)
|
|
if end != -1:
|
|
text = re.sub(r"[^\n]", " ", text[: end + 4]) + text[end + 4:]
|
|
steps = [
|
|
r"(?ms)^\s*```.*?^\s*```", # code blocks
|
|
r"(?s)<!--.*?-->", # comments
|
|
r"(?m)^\s*>.*$", # block quotes
|
|
r"`[^`\n]+`", # code spans
|
|
r"\]\([^)\s]*\)", # link targets
|
|
r"~~[^~\n]+~~", # struck through: a word named as retired
|
|
r"\"[^\"\n]*\"", # "quoted"
|
|
r"\u201c[^\u201d]*\u201d", # curly double quotes
|
|
r"\u2018[^\u2019\n]*\u2019", # curly single quotes
|
|
]
|
|
if glossary_file:
|
|
steps[:0] = [NOT_LINE.pattern, IDENT_LINE.pattern]
|
|
for step in steps:
|
|
text = re.sub(step, blank, text)
|
|
return text
|
|
|
|
|
|
def research_and_issues():
|
|
"""The research efforts and issue reports dated on or after FROM, file by file."""
|
|
out = []
|
|
for folder, key, first in (("01-RESEARCH", "initiated", "00-overview.md"), ("04-ISSUES", "opened", "00-report.md")):
|
|
base = os.path.join(ROOT, folder)
|
|
for effort in sorted(os.listdir(base)):
|
|
head = os.path.join(base, effort, first)
|
|
if not os.path.isfile(head):
|
|
continue
|
|
date = frontmatter_field(read(rel(head)), key)
|
|
if not date or date < FROM:
|
|
continue
|
|
for name in sorted(os.listdir(os.path.join(base, effort))):
|
|
if name.endswith(".md"):
|
|
out.append("%s/%s/%s" % (folder, effort, name))
|
|
return out
|
|
|
|
|
|
def checked_documents():
|
|
docs = []
|
|
for entry in ALWAYS:
|
|
path = os.path.join(ROOT, entry)
|
|
if os.path.isfile(path):
|
|
docs.append(entry)
|
|
continue
|
|
for base, dirs, files in os.walk(path):
|
|
dirs.sort()
|
|
for name in sorted(files):
|
|
if name.endswith(".md"):
|
|
docs.append(rel(os.path.join(base, name)))
|
|
return docs + research_and_issues()
|
|
|
|
|
|
def allowed():
|
|
"""words-allowed.md: one row per document — | `document` | until | why |.
|
|
|
|
`until` is a date, after which the allowance fails like the words themselves, or `kept` for a
|
|
document that records what was said and must keep its words: only a research effort that has
|
|
graduated may be kept, because once it has become a decision it is a record like one.
|
|
"""
|
|
out, problems = {}, []
|
|
if not os.path.isfile(os.path.join(ROOT, ALLOWED)):
|
|
return out, problems
|
|
today = datetime.date.today().isoformat()
|
|
for line in read(ALLOWED).splitlines():
|
|
cells = [c.strip() for c in line.strip().strip("|").split("|")]
|
|
if len(cells) != 3 or not cells[0].startswith("`"):
|
|
continue
|
|
doc, until, why = cells[0].strip("`"), cells[1], cells[2]
|
|
if not why:
|
|
problems.append("%s: the allowance for %s gives no reason" % (ALLOWED, doc))
|
|
if until == "kept":
|
|
overview = os.path.join(os.path.dirname(doc), "00-overview.md")
|
|
status = frontmatter_field(read(overview), "status") if os.path.isfile(os.path.join(ROOT, overview)) else None
|
|
if not doc.startswith("01-RESEARCH/") or status != "graduated":
|
|
problems.append("%s: %s is kept, but only a graduated research effort may be" % (ALLOWED, doc))
|
|
continue
|
|
elif not re.match(r"^\d{4}-\d{2}-\d{2}$", until):
|
|
problems.append("%s: the allowance for %s has neither a date nor `kept`" % (ALLOWED, doc))
|
|
continue
|
|
elif until < today:
|
|
problems.append("%s: the allowance for %s ran out on %s — reword it" % (ALLOWED, doc, until))
|
|
continue
|
|
out[doc] = until
|
|
return out, problems
|
|
|
|
|
|
def check_retired(retired, identifiers):
|
|
problems = []
|
|
allowance, problems_allowed = allowed()
|
|
problems += problems_allowed
|
|
words = [(w, pattern(w), "retired") for w, scope in retired if scope in ("all", "hq")]
|
|
words += [(i, pattern(i), "an identifier, allowed only in a code span") for i in identifiers]
|
|
used = set()
|
|
for doc in checked_documents():
|
|
text = prose(read(doc), glossary_file=(doc == GLOSSARY))
|
|
for word, rx, why in words:
|
|
for m in rx.finditer(text):
|
|
if doc in allowance:
|
|
used.add(doc)
|
|
continue
|
|
line = text.count("\n", 0, m.start()) + 1
|
|
problems.append("%s:%d: %r is %s (glossary)" % (doc, line, m.group(0).replace("\n", " "), why))
|
|
for doc in allowance:
|
|
if doc not in used:
|
|
problems.append("%s: %s no longer uses a retired word — remove its allowance" % (ALLOWED, doc))
|
|
return problems
|
|
|
|
|
|
def check_unique(retired, heads):
|
|
problems = []
|
|
seen = {}
|
|
for head in heads:
|
|
key = head.lower()
|
|
if key in seen:
|
|
problems.append("%s: %r heads two entries" % (GLOSSARY, head))
|
|
seen[key] = True
|
|
for word, _ in retired:
|
|
if word.lower() in seen:
|
|
problems.append("%s: %r is a head word and also retired" % (GLOSSARY, word))
|
|
return problems
|
|
|
|
|
|
def tools_list(retired):
|
|
return sorted({w for w, scope in retired if scope in ("all", "tools")}, key=str.lower)
|
|
|
|
|
|
def check_copy(retired):
|
|
catalogue = os.environ.get("MESH_CATALOG_DIR")
|
|
if not catalogue:
|
|
print("NOT COMPARED: MESH_CATALOG_DIR is not set, so the catalogue's copy of the list was not read")
|
|
return []
|
|
path = os.path.join(catalogue, "retired-words")
|
|
try:
|
|
with open(path, encoding="utf-8") as handle:
|
|
copy = [l.strip() for l in handle if l.strip() and not l.startswith("#")]
|
|
except OSError as err:
|
|
return ["the catalogue's copy of the retired words cannot be read: %s" % err]
|
|
want = tools_list(retired)
|
|
if sorted(copy, key=str.lower) != want:
|
|
return ["the catalogue's retired-words differs from the glossary: it should list %s" % ", ".join(want)]
|
|
return []
|
|
|
|
|
|
def main(argv):
|
|
retired, identifiers, heads = glossary()
|
|
if argv[1:2] == ["--list"]:
|
|
scope = argv[2] if len(argv) > 2 else "tools"
|
|
words = tools_list(retired) if scope == "tools" else sorted({w for w, s in retired if s in ("all", scope)})
|
|
print("\n".join(words))
|
|
return 0
|
|
if not retired:
|
|
print("words: the glossary names no retired word on a *Not:* line — the check would pass on anything")
|
|
return 1
|
|
problems = check_unique(retired, heads) + check_retired(retired, identifiers) + check_copy(retired)
|
|
for p in problems:
|
|
print(p)
|
|
if problems:
|
|
print("words: %d problem(s)" % len(problems))
|
|
return 1
|
|
print("words: %d retired words and %d identifiers, none in running prose; %d head words, each once"
|
|
% (len(retired), len(identifiers), len(heads)))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main(sys.argv))
|