package facts import ( "crypto/sha256" "fmt" "net" "regexp" "sort" "strings" ) // Withheld is what a value that reads as a secret becomes. A check composing with it gets a stand-in of // the right kind (a string), never the value. const Withheld = "withheld" // Pseudonym is the stable stand-in for a name: the same length, letters for letters and digits for // digits, anything else kept where it was. Stable, so two snapshots of one mesh name a machine alike and // a check's verdict can be compared across them; derived from the name alone, so it needs no key to be // kept anywhere. // // It hides nothing from somebody who can guess the names: that is not its purpose. Its purpose is that a // snapshot — and every fixture, replay or pull-request comment made from one — carries no name of the // installation it came from, while every length a limit meets stays the length it was. func Pseudonym(kind, name string) string { sum := sha256.Sum256([]byte("novox-mesh-facts\x00" + kind + "\x00" + name)) out := []byte(name) for i, c := range out { b := sum[i%len(sum)] ^ byte(i/len(sum)) switch { case c >= 'a' && c <= 'z', c >= 'A' && c <= 'Z': out[i] = 'a' + b%26 case c >= '0' && c <= '9': out[i] = '0' + b%10 } } // A name must still read as one: a machine's begins with a letter. if len(out) > 0 && (out[0] < 'a' || out[0] > 'z') && name[0] >= 'a' && name[0] <= 'z' { out[0] = 'a' + sum[0]%26 } return string(out) } // Scrubber replaces what a snapshot must not carry, wherever it appears in text. type Scrubber struct { // replace is every real word and what it becomes, longest first, so a domain is replaced before a // label inside it. replace [][2]string seen map[string]bool } // NewScrubber knows the installation's names: machines, sites, accounts, public domains. func NewScrubber() *Scrubber { return &Scrubber{seen: map[string]bool{}} } // Machine registers a machine's name and answers its pseudonym. func (s *Scrubber) Machine(name string) string { return s.word("machine", name) } // Site registers a site's name and answers its pseudonym. func (s *Scrubber) Site(name string) string { return s.word("site", name) } // Account registers an account's name and answers its pseudonym. func (s *Scrubber) Account(name string) string { return s.word("account", name) } // Domain registers a public domain and answers its stand-in: each label but the last replaced, so the // shape and every length stay. func (s *Scrubber) Domain(domain string) string { if domain == "" { return "" } labels := strings.Split(domain, ".") for i := range labels { if i == len(labels)-1 && len(labels) > 1 { break } labels[i] = Pseudonym("domain", labels[i]) } out := strings.Join(labels, ".") s.add(domain, out) return out } func (s *Scrubber) word(kind, name string) string { if name == "" { return "" } out := Pseudonym(kind, name) s.add(name, out) return out } func (s *Scrubber) add(from, to string) { if from == "" || s.seen[from] { return } s.seen[from] = true s.replace = append(s.replace, [2]string{from, to}) sort.SliceStable(s.replace, func(i, j int) bool { return len(s.replace[i][0]) > len(s.replace[j][0]) }) } // wordBoundary is what may stand beside a name for it to be that name and not part of another word. func wordBoundary(c byte) bool { return !(c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' || c >= '0' && c <= '9' || c == '_') } // Text is s with every known name replaced, every address put in a documentation range, every address // of mail replaced, and every run that reads as a secret withheld. func (s *Scrubber) Text(text string) string { if text == "" { return text } text = secretRuns(text) text = emails.ReplaceAllStringFunc(text, func(mail string) string { if strings.HasPrefix(mail, Withheld+"@") { return mail // a URL's withheld credentials, not an address of mail } return "someone@example.org" }) text = addresses(text) for _, r := range s.replace { text = replaceWord(text, r[0], r[1]) } return text } // replaceWord replaces from where it stands as a word of its own. func replaceWord(text, from, to string) string { var b strings.Builder for { i := strings.Index(text, from) if i < 0 { b.WriteString(text) return b.String() } end := i + len(from) if (i == 0 || wordBoundary(text[i-1])) && (end == len(text) || wordBoundary(text[end])) { b.WriteString(text[:i]) b.WriteString(to) } else { b.WriteString(text[:end]) } text = text[end:] } } // Values is a settings layer with every key that names a secret withheld, and every string scrubbed. func (s *Scrubber) Values(values map[string]any) map[string]any { out := make(map[string]any, len(values)) for k, v := range values { // A key may itself be a name — a map of machines to something. key := s.Text(k) if secretKey.MatchString(k) { out[key] = withheldLike(v) continue } out[key] = s.value(v) } return out } func (s *Scrubber) value(v any) any { switch t := v.(type) { case string: return s.Text(t) case map[string]any: return s.Values(t) case []any: out := make([]any, len(t)) for i, e := range t { out[i] = s.value(e) } return out default: return v } } // withheldLike is a stand-in of the same kind: a string for a string, a list for a list, so a check // composing a setting still finds the shape it expects. func withheldLike(v any) any { switch t := v.(type) { case nil: return nil case bool, float64, int, int64: return t case []any: out := make([]any, len(t)) for i, e := range t { out[i] = withheldLike(e) } return out case map[string]any: out := map[string]any{} for k, e := range t { out[k] = withheldLike(e) } return out default: return Withheld } } // secretKey is a setting's key that names a secret. var secretKey = regexp.MustCompile(`(?i)(secret|passw|token|api[-_]?key|private[-_]?key|credential|bearer|cookie|salt|signing)`) // emails are addresses of mail: a person's name, or an installation's domain, in either half. var emails = regexp.MustCompile(`[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}`) // secretRun is a run of characters a key, a token or a hash is made of, long enough to be one. var secretRun = regexp.MustCompile(`[A-Za-z0-9+/=_\-]{24,}`) // userinfo is the credentials part of a URL. var userinfo = regexp.MustCompile(`(://)[^/@\s:]+:[^/@\s]+@`) // secretRuns withholds a URL's credentials and every run that reads as a key: long, and mixing letters // with digits or symbols. A long plain word (a path, a module's name) is left alone. func secretRuns(text string) string { if strings.Contains(text, "-----BEGIN") { return Withheld } text = userinfo.ReplaceAllString(text, "${1}"+Withheld+"@") var b strings.Builder last := 0 for _, at := range secretRun.FindAllStringIndex(text, -1) { run := text[at[0]:at[1]] b.WriteString(text[last:at[0]]) last = at[1] // A digest names an artifact, not a secret. if strings.HasSuffix(text[:at[0]], "sha256:") { b.WriteString(run) continue } judged := judgeRun(run) // A path withheld keeps a path's shape — an absolute one stays absolute — so a check composing a // place or an access still finds a path where one must be. if judged == Withheld && strings.HasPrefix(run, "/") { judged = "/" + Withheld } b.WriteString(judged) } b.WriteString(text[last:]) return b.String() } // judgeRun is a run withheld when it reads as a key: long, and mixing letters with digits or symbols. func judgeRun(run string) string { { var lower, upper, digit, symbol bool for _, c := range run { switch { case c >= 'a' && c <= 'z': lower = true case c >= 'A' && c <= 'Z': upper = true case c >= '0' && c <= '9': digit = true default: symbol = true } } classes := 0 for _, b := range []bool{lower, upper, digit, symbol} { if b { classes++ } } // A sha256 digest names an artifact, not a secret, and a dashed word is a name. if strings.HasPrefix(run, "sha256") || (!digit && !upper) { return run } if classes >= 3 || (digit && (lower || upper) && len(run) >= 32) { return Withheld } return run } } // addresses puts every IP address in text into a documentation range (RFC 5737, RFC 3849): the same // address always becomes the same stand-in, so two settings naming one machine still name one. func addresses(text string) string { var b strings.Builder i := 0 for i < len(text) { if !addressChar(text[i]) { b.WriteByte(text[i]) i++ continue } j := i for j < len(text) && addressChar(text[j]) { j++ } b.WriteString(standIn(text[i:j])) i = j } return b.String() } func addressChar(c byte) bool { return c >= '0' && c <= '9' || c >= 'a' && c <= 'f' || c >= 'A' && c <= 'F' || c == '.' || c == ':' } // standIn is the documentation address for a run that is an address, or the run itself. func standIn(run string) string { trimmed := strings.TrimRight(run, ".:") tail := run[len(trimmed):] ip := net.ParseIP(trimmed) switch { case ip == nil: // A port after an IPv4 address reads as part of the run: try without it. if host, port, ok := strings.Cut(trimmed, ":"); ok && strings.Count(trimmed, ":") == 1 && net.ParseIP(host) != nil && strings.Contains(host, ".") { return standIn(host) + ":" + port + tail } return run case ip.To4() != nil && strings.Contains(trimmed, "."): sum := sha256.Sum256([]byte("v4\x00" + trimmed)) ranges := []string{"192.0.2", "198.51.100", "203.0.113"} return fmt.Sprintf("%s.%d", ranges[sum[0]%3], 1+sum[1]%254) + tail case strings.Count(trimmed, ":") >= 2: sum := sha256.Sum256([]byte("v6\x00" + trimmed)) return fmt.Sprintf("2001:db8::%x:%x", uint16(sum[0])<<8|uint16(sum[1]), uint16(sum[2])<<8|uint16(sum[3])) + tail } return run } // RepositoryName is a repository as `owner/repository`, without the forge's address it was cloned from: // http://forge.internal:3000/owner/repo.git → owner/repo (novox/hq issue 288). What a check matches a // pull request's repository by is its owner and name, never the forge's address, so nothing is lost. func RepositoryName(repository string) string { r := strings.TrimSuffix(strings.TrimSuffix(strings.TrimSpace(repository), "/"), ".git") if r == "" { return "" } if _, rest, found := strings.Cut(r, "://"); found { r = rest if _, path, found := strings.Cut(r, "/"); found { r = path } else { return "" } } else if at := strings.Index(r, "@"); at >= 0 { // scp-like: git@forge:owner/repo if _, path, found := strings.Cut(r[at+1:], ":"); found { r = path } } parts := strings.Split(strings.Trim(r, "/"), "/") if len(parts) >= 2 { return parts[len(parts)-2] + "/" + parts[len(parts)-1] } return parts[len(parts)-1] }