An agent meets the mesh's words most often in tool descriptions, and nothing compared them with the glossary: several still said "the host" for the node-engine and the forge's pull request comment was headed "Change plan", a word retired twice over. retired-words is the copy of the words the glossary retires for the tools, and checks/words fails the repository check when any string a module's code can show, or any manifest description, uses one. Those found are reworded here.
328 lines
8.1 KiB
Go
328 lines
8.1 KiB
Go
// Command words holds what the catalogue's modules say to an agent to the glossary's words (novox/hq ADR
|
|
// 0244): no word the glossary retired for the tools' descriptions — the copy in retired-words at the
|
|
// catalogue's root — in any text a module's tools can show. That text is every string literal in a module's
|
|
// own code that is not a test (a tool's description, the notes and errors it answers with) and every
|
|
// description in its manifest. Comments are not read: an agent never sees them.
|
|
//
|
|
// A word is matched whole and in any case, a space in it matching any run of white space. A vendor word is
|
|
// never on the list (the glossary does not retire one for the tools), so a wrapped program's own objects —
|
|
// an identity provider's users, a media manager's releases — are never findings.
|
|
//
|
|
// go run . <catalogue root>
|
|
//
|
|
// Exits 1 on a finding, 2 when the list cannot be read.
|
|
package main
|
|
|
|
import (
|
|
"bufio"
|
|
"encoding/json"
|
|
"fmt"
|
|
"go/scanner"
|
|
"go/token"
|
|
"io/fs"
|
|
"os"
|
|
"path/filepath"
|
|
"regexp"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
)
|
|
|
|
// text is one piece of text an agent can be shown, and where it starts.
|
|
type text struct {
|
|
file string
|
|
line int
|
|
s string
|
|
}
|
|
|
|
func main() {
|
|
root := "."
|
|
if len(os.Args) > 1 {
|
|
root = os.Args[1]
|
|
}
|
|
words, err := readList(filepath.Join(root, "retired-words"))
|
|
if err != nil {
|
|
fmt.Fprintln(os.Stderr, "words:", err)
|
|
os.Exit(2)
|
|
}
|
|
if len(words) == 0 {
|
|
fmt.Fprintln(os.Stderr, "words: retired-words lists no word — the check would pass on anything")
|
|
os.Exit(2)
|
|
}
|
|
var texts []text
|
|
files := 0
|
|
err = filepath.WalkDir(filepath.Join(root, "modules"), func(path string, d fs.DirEntry, err error) error {
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if d.IsDir() {
|
|
switch d.Name() {
|
|
case "node_modules", "dist", "vendor", "test", "tests", "testdata", ".git":
|
|
return filepath.SkipDir
|
|
}
|
|
return nil
|
|
}
|
|
rel, _ := filepath.Rel(root, path)
|
|
name := d.Name()
|
|
var found []text
|
|
switch {
|
|
case strings.HasSuffix(name, "_test.go"), strings.Contains(name, ".test."), strings.Contains(name, ".spec."):
|
|
return nil
|
|
case strings.HasSuffix(name, ".go"):
|
|
found, err = goStrings(path, rel)
|
|
case strings.HasSuffix(name, ".ts"), strings.HasSuffix(name, ".js"), strings.HasSuffix(name, ".mjs"):
|
|
found, err = scriptStrings(path, rel)
|
|
case name == "module.json":
|
|
found, err = jsonStrings(path, rel)
|
|
default:
|
|
return nil
|
|
}
|
|
if err != nil {
|
|
return fmt.Errorf("%s: %w", rel, err)
|
|
}
|
|
files++
|
|
texts = append(texts, found...)
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
fmt.Fprintln(os.Stderr, "words:", err)
|
|
os.Exit(2)
|
|
}
|
|
findings := check(words, texts)
|
|
for _, f := range findings {
|
|
fmt.Println(f)
|
|
}
|
|
if len(findings) > 0 {
|
|
fmt.Printf("words: %d use(s) of a word the glossary retired (novox/hq ADR 0244) — say it in the glossary's word\n", len(findings))
|
|
os.Exit(1)
|
|
}
|
|
fmt.Printf("words: %d retired words, none in the %d strings of %d files\n", len(words), len(texts), files)
|
|
}
|
|
|
|
// readList reads retired-words: one word or phrase a line; blank lines and lines starting with # are not words.
|
|
func readList(path string) ([]string, error) {
|
|
f, err := os.Open(path)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer f.Close()
|
|
var words []string
|
|
sc := bufio.NewScanner(f)
|
|
for sc.Scan() {
|
|
line := strings.TrimSpace(sc.Text())
|
|
if line == "" || strings.HasPrefix(line, "#") {
|
|
continue
|
|
}
|
|
words = append(words, line)
|
|
}
|
|
return words, sc.Err()
|
|
}
|
|
|
|
func pattern(word string) *regexp.Regexp {
|
|
parts := strings.Fields(word)
|
|
for i, p := range parts {
|
|
parts[i] = regexp.QuoteMeta(p)
|
|
}
|
|
return regexp.MustCompile(`(?i)(?:^|[^\w-])(` + strings.Join(parts, `\s+`) + `)(?:$|[^\w-])`)
|
|
}
|
|
|
|
func check(words []string, texts []text) []string {
|
|
var out []string
|
|
for _, w := range words {
|
|
rx := pattern(w)
|
|
for _, t := range texts {
|
|
if m := rx.FindStringSubmatchIndex(t.s); m != nil {
|
|
line := t.line
|
|
if line > 0 {
|
|
line += strings.Count(t.s[:m[2]], "\n")
|
|
}
|
|
out = append(out, fmt.Sprintf("%s:%d: %q is retired — in %q", t.file, line, w, excerpt(t.s, m[2], m[3])))
|
|
}
|
|
}
|
|
}
|
|
sort.Strings(out)
|
|
return out
|
|
}
|
|
|
|
func excerpt(s string, from, to int) string {
|
|
a, b := from-50, to+50
|
|
if a < 0 {
|
|
a = 0
|
|
}
|
|
if b > len(s) {
|
|
b = len(s)
|
|
}
|
|
return strings.Join(strings.Fields(s[a:b]), " ")
|
|
}
|
|
|
|
// goStrings returns a Go file's string literals, adjacent ones joined across `+` as the program joins them.
|
|
func goStrings(path, rel string) ([]text, error) {
|
|
src, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
fset := token.NewFileSet()
|
|
file := fset.AddFile(rel, -1, len(src))
|
|
var s scanner.Scanner
|
|
var bad error
|
|
s.Init(file, src, func(pos token.Position, msg string) { bad = fmt.Errorf("%s: %s", pos, msg) }, 0)
|
|
var out []text
|
|
var cur *text
|
|
joining := false
|
|
for {
|
|
pos, tok, lit := s.Scan()
|
|
if tok == token.EOF {
|
|
break
|
|
}
|
|
switch {
|
|
case tok == token.STRING:
|
|
v, err := strconv.Unquote(lit)
|
|
if err != nil {
|
|
v = lit
|
|
}
|
|
if cur != nil && joining {
|
|
cur.s += v
|
|
} else {
|
|
out = append(out, text{file: rel, line: fset.Position(pos).Line, s: v})
|
|
cur = &out[len(out)-1]
|
|
}
|
|
joining = false
|
|
case tok == token.ADD && cur != nil:
|
|
joining = true
|
|
default:
|
|
cur, joining = nil, false
|
|
}
|
|
}
|
|
return out, bad
|
|
}
|
|
|
|
// scriptStrings returns a TypeScript or JavaScript file's string literals, comments skipped, adjacent ones
|
|
// joined across `+`. A template literal's `${…}` parts are left out of the text around them.
|
|
func scriptStrings(path, rel string) ([]text, error) {
|
|
b, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
src := string(b)
|
|
var out []text
|
|
line := 1
|
|
lastWasString, joining := false, false
|
|
for i := 0; i < len(src); i++ {
|
|
c := src[i]
|
|
switch {
|
|
case c == '\n':
|
|
line++
|
|
case c == '/' && i+1 < len(src) && src[i+1] == '/':
|
|
for i < len(src) && src[i] != '\n' {
|
|
i++
|
|
}
|
|
line++
|
|
case c == '/' && i+1 < len(src) && src[i+1] == '*':
|
|
end := strings.Index(src[i+2:], "*/")
|
|
if end < 0 {
|
|
end = len(src) - i - 2
|
|
}
|
|
line += strings.Count(src[i:i+2+end], "\n")
|
|
i += end + 3
|
|
case c == '"' || c == '\'' || c == '`':
|
|
start := line
|
|
var sb strings.Builder
|
|
depth := 0
|
|
j := i + 1
|
|
for ; j < len(src); j++ {
|
|
d := src[j]
|
|
if d == '\n' {
|
|
line++
|
|
}
|
|
if depth > 0 {
|
|
if d == '\n' {
|
|
sb.WriteByte('\n')
|
|
}
|
|
if d == '{' {
|
|
depth++
|
|
} else if d == '}' {
|
|
depth--
|
|
}
|
|
continue
|
|
}
|
|
if d == '\\' && j+1 < len(src) {
|
|
j++
|
|
sb.WriteByte(src[j])
|
|
continue
|
|
}
|
|
if c == '`' && d == '$' && j+1 < len(src) && src[j+1] == '{' {
|
|
depth = 1
|
|
j++
|
|
sb.WriteByte(' ')
|
|
continue
|
|
}
|
|
if d == c || (c != '`' && d == '\n') {
|
|
break
|
|
}
|
|
sb.WriteByte(d)
|
|
}
|
|
if lastWasString && joining && len(out) > 0 {
|
|
out[len(out)-1].s += sb.String()
|
|
} else {
|
|
out = append(out, text{file: rel, line: start, s: sb.String()})
|
|
}
|
|
lastWasString, joining = true, false
|
|
i = j
|
|
continue
|
|
case c == '+' && lastWasString:
|
|
joining = true
|
|
continue
|
|
case c == ' ' || c == '\t' || c == '\r':
|
|
continue
|
|
default:
|
|
lastWasString, joining = false, false
|
|
}
|
|
if c == '\n' {
|
|
continue
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// jsonStrings returns the descriptions in a manifest — every string under a key named "description", at any
|
|
// depth. The rest of a manifest is names, paths and the contents of files it places, none of which a tool
|
|
// shows an agent.
|
|
func jsonStrings(path, rel string) ([]text, error) {
|
|
b, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var v any
|
|
if err := json.Unmarshal(b, &v); err != nil {
|
|
return nil, err
|
|
}
|
|
var out []text
|
|
var walk func(any, bool)
|
|
walk = func(v any, described bool) {
|
|
switch x := v.(type) {
|
|
case string:
|
|
if described {
|
|
out = append(out, text{file: rel, line: lineOf(string(b), x), s: x})
|
|
}
|
|
case []any:
|
|
for _, e := range x {
|
|
walk(e, described)
|
|
}
|
|
case map[string]any:
|
|
for k, e := range x {
|
|
walk(e, k == "description")
|
|
}
|
|
}
|
|
}
|
|
walk(v, false)
|
|
return out, nil
|
|
}
|
|
|
|
func lineOf(src, s string) int {
|
|
q, _ := json.Marshal(s)
|
|
if i := strings.Index(src, string(q)); i >= 0 {
|
|
return strings.Count(src[:i], "\n") + 1
|
|
}
|
|
return 0
|
|
}
|