aboutsummaryrefslogtreecommitdiff
path: root/internal/ignore/ignore.go
diff options
context:
space:
mode:
Diffstat (limited to 'internal/ignore/ignore.go')
-rw-r--r--internal/ignore/ignore.go324
1 files changed, 324 insertions, 0 deletions
diff --git a/internal/ignore/ignore.go b/internal/ignore/ignore.go
new file mode 100644
index 0000000..fff6dcc
--- /dev/null
+++ b/internal/ignore/ignore.go
@@ -0,0 +1,324 @@
+// SPDX-License-Identifier: GPL-3.0-or-later
+
+// Package ignore implements a gitignore(5)-compatible path matcher.
+package ignore
+
+import (
+ "errors"
+ "fmt"
+ "regexp"
+ "strings"
+)
+
+// pattern is one compiled gitignore(5) pattern.
+type pattern struct {
+ re *regexp.Regexp
+ negate bool
+ dirOnly bool
+}
+
+// Matcher decides which paths a set of gitignore(5) patterns ignores.
+type Matcher struct {
+ patterns []pattern
+}
+
+// New compiles patterns in gitignore(5) syntax, in order. Blank patterns and
+// patterns starting with "#" are skipped. A nil or empty list gives a
+// matcher that matches nothing.
+func New(patterns []string) (*Matcher, error) {
+ m := &Matcher{}
+ for _, orig := range patterns {
+ p := orig
+ if p == "" || strings.HasPrefix(p, "#") {
+ continue
+ }
+ negate := false
+ switch {
+ case strings.HasPrefix(p, "!"):
+ negate = true
+ p = p[1:]
+ case strings.HasPrefix(p, `\!`), strings.HasPrefix(p, `\#`):
+ p = p[1:]
+ }
+ p = trimTrailingSpaces(p)
+ dirOnly := false
+ if strings.HasSuffix(p, "/") {
+ dirOnly = true
+ p = strings.TrimSuffix(p, "/")
+ }
+ reStr, err := toRegexp(p)
+ if err != nil {
+ return nil, fmt.Errorf("bad ignore pattern %q: %v", orig, err)
+ }
+ re, err := regexp.Compile(reStr)
+ if err != nil {
+ return nil, fmt.Errorf("bad ignore pattern %q: %v", orig, err)
+ }
+ m.patterns = append(m.patterns, pattern{re: re, negate: negate, dirOnly: dirOnly})
+ }
+ return m, nil
+}
+
+// trimTrailingSpaces removes trailing spaces that are not escaped by a
+// preceding backslash, per gitignore(5).
+func trimTrailingSpaces(p string) string {
+ for strings.HasSuffix(p, " ") {
+ n := 0
+ for i := len(p) - 2; i >= 0 && p[i] == '\\'; i-- {
+ n++
+ }
+ if n%2 == 1 {
+ break // the trailing space is escaped: keep it
+ }
+ p = p[:len(p)-1]
+ }
+ return p
+}
+
+// Match reports whether rel is ignored. rel is slash-separated and relative
+// to the root, with no leading slash and no trailing slash, and each of its
+// path components must be "." and ".."-free (as produced by a directory
+// walk, not a raw user-typed path).
+func (m *Matcher) Match(rel string, isDir bool) bool {
+ parts := strings.Split(rel, "/")
+ for i := 1; i < len(parts); i++ {
+ ancestor := strings.Join(parts[:i], "/")
+ if m.matchOne(ancestor, true) {
+ return true
+ }
+ }
+ return m.matchOne(rel, isDir)
+}
+
+// matchOne applies every pattern in order to one path and returns the final
+// ignored state, without considering ancestor directories.
+func (m *Matcher) matchOne(path string, isDir bool) bool {
+ ignored := false
+ for _, pat := range m.patterns {
+ if pat.dirOnly && !isDir {
+ continue
+ }
+ if pat.re.MatchString(path) {
+ ignored = !pat.negate
+ }
+ }
+ return ignored
+}
+
+// toRegexp translates the body of one gitignore pattern (no leading "!",
+// no trailing "/") into a regular expression over slash-separated paths.
+func toRegexp(p string) (string, error) {
+ anchored := strings.Contains(p, "/")
+ p = strings.TrimPrefix(p, "/")
+ var b strings.Builder
+ b.WriteString("^")
+ if !anchored {
+ b.WriteString("(?:.*/)?")
+ }
+ for i := 0; i < len(p); i++ {
+ c := p[i]
+ switch {
+ case strings.HasPrefix(p[i:], "**/") && (i == 0 || p[i-1] == '/'):
+ b.WriteString("(?:.*/)?")
+ i += 2
+ case strings.HasPrefix(p[i:], "/**") && i+3 == len(p):
+ b.WriteString("/.*")
+ i += 2
+ case strings.HasPrefix(p[i:], "**"):
+ b.WriteString("[^/]*")
+ i++
+ case c == '*':
+ b.WriteString("[^/]*")
+ case c == '?':
+ b.WriteString("[^/]")
+ case c == '[':
+ negated := false
+ start := i + 1
+ if start < len(p) && (p[start] == '!' || p[start] == '^') {
+ negated = true
+ start++
+ }
+ items, end, err := parseClassBody(p, start)
+ if err != nil {
+ return "", err
+ }
+ b.WriteString(renderClass(items, negated))
+ i = end
+ case c == '\\' && i+1 < len(p):
+ i++
+ b.WriteString(regexp.QuoteMeta(string(p[i])))
+ default:
+ b.WriteString(regexp.QuoteMeta(string(c)))
+ }
+ }
+ b.WriteString("$")
+ return b.String(), nil
+}
+
+// classItem is one member of a gitignore bracket expression: either a
+// verbatim POSIX bracket subexpression ("[:digit:]", "[.ch.]", "[=e=]"), or
+// a literal byte (lo == hi) or a byte range (lo-hi).
+type classItem struct {
+ posix string
+ lo, hi byte
+}
+
+// parseClassBody scans p starting at start (just past "[", "[!" or "[^")
+// for the members of a bracket expression, up to and including the
+// matching unescaped "]". A "]" appearing immediately at start is a
+// literal member, not the terminator, per glob syntax (e.g. "[]a]" matches
+// "]" or "a"). Inside the class, "\x" escapes x to a literal member,
+// neutralizing any meaning x would otherwise have (closing the class,
+// starting a POSIX subexpression, forming a range) — matching git's
+// wildmatch. It returns the members found and the index of the closing
+// "]", or an error if none is found (including a class left unterminated
+// because its only "]" was escaped).
+func parseClassBody(p string, start int) ([]classItem, int, error) {
+ var items []classItem
+ k := start
+ if k < len(p) && p[k] == ']' {
+ items = append(items, classItem{lo: ']', hi: ']'})
+ k++
+ }
+ for k < len(p) && p[k] != ']' {
+ // "\x" escapes x to a literal member; it can never start a
+ // POSIX subexpression or a range. A backslash with nothing
+ // after it can never close the class either.
+ if p[k] == '\\' {
+ if k+1 >= len(p) {
+ return nil, 0, errors.New("unterminated [")
+ }
+ items = append(items, classItem{lo: p[k+1], hi: p[k+1]})
+ k += 2
+ continue
+ }
+ if p[k] == '[' && k+1 < len(p) && (p[k+1] == ':' || p[k+1] == '.' || p[k+1] == '=') {
+ delim := p[k+1]
+ rest := strings.Index(p[k+2:], string(delim)+"]")
+ if rest < 0 {
+ return nil, 0, errors.New("unterminated [")
+ }
+ stop := k + 2 + rest + 2
+ items = append(items, classItem{posix: p[k:stop]})
+ k = stop
+ continue
+ }
+ // "x-y" is a range unless the "-" is the last character before
+ // the closing "]", in which case it is a literal "-".
+ if k+2 < len(p) && p[k+1] == '-' && p[k+2] != ']' {
+ items = append(items, classItem{lo: p[k], hi: p[k+2]})
+ k += 3
+ continue
+ }
+ items = append(items, classItem{lo: p[k], hi: p[k]})
+ k++
+ }
+ if k >= len(p) {
+ return nil, 0, errors.New("unterminated [")
+ }
+ return items, k, nil
+}
+
+// renderClass turns the members of a bracket expression into a Go regexp
+// character class. Git's bracket expressions never match "/", regardless
+// of negation, so "/" is dropped from a positive class (splitting any range
+// that spans it) and added to a negated one. A positive class left
+// matching nothing (e.g. "[/]") is rendered as a class that can never
+// match any character, since RE2 has no empty class or lookahead.
+func renderClass(items []classItem, negated bool) string {
+ if negated {
+ items = append(items, classItem{lo: '/', hi: '/'})
+ return "[^" + renderClassItems(items) + "]"
+ }
+ items = excludeSlash(items)
+ if len(items) == 0 {
+ return `[^\x00-\x{10FFFF}]`
+ }
+ return "[" + renderClassItems(items) + "]"
+}
+
+// posixSlashFree gives the ASCII definition, minus "/", of every standard
+// POSIX bracket class whose definition otherwise includes it: graph and
+// print (visible characters, "/" among them) and punct (documented in
+// gitignore's own toRegexp derivation as "!-. :-@ [-` {-~", i.e. punct's
+// usual "!-/" range with "/" trimmed to "!-."). The other nine standard
+// classes (alnum, alpha, blank, cntrl, digit, lower, space, upper,
+// xdigit) never contain "/" and so need no rewriting.
+var posixSlashFree = map[string][]classItem{
+ "graph": {{lo: '!', hi: '.'}, {lo: '0', hi: '~'}},
+ "print": {{lo: ' ', hi: '.'}, {lo: '0', hi: '~'}},
+ "punct": {{lo: '!', hi: '.'}, {lo: ':', hi: '@'}, {lo: '[', hi: '`'}, {lo: '{', hi: '~'}},
+}
+
+// posixClassName returns the name inside a "[:name:]" POSIX bracket class,
+// or "", false for anything else (a collating symbol "[.x.]", an
+// equivalence class "[=x=]", or a malformed value).
+func posixClassName(posix string) (string, bool) {
+ if strings.HasPrefix(posix, "[:") && strings.HasSuffix(posix, ":]") {
+ return posix[2 : len(posix)-2], true
+ }
+ return "", false
+}
+
+// excludeSlash removes "/" from a positive class's members, splitting any
+// range that spans it into the parts on either side, and substituting the
+// "/"-free ASCII definition for any POSIX class that would otherwise
+// include it.
+func excludeSlash(items []classItem) []classItem {
+ var out []classItem
+ for _, it := range items {
+ switch {
+ case it.posix != "":
+ if name, ok := posixClassName(it.posix); ok {
+ if repl, hasSlash := posixSlashFree[name]; hasSlash {
+ out = append(out, repl...)
+ continue
+ }
+ }
+ out = append(out, it)
+ case it.lo == '/' && it.hi == '/':
+ // drop the lone literal "/"
+ case it.lo <= '/' && '/' <= it.hi:
+ if it.lo <= '/'-1 {
+ out = append(out, classItem{lo: it.lo, hi: '/' - 1})
+ }
+ if '/'+1 <= it.hi {
+ out = append(out, classItem{lo: '/' + 1, hi: it.hi})
+ }
+ default:
+ out = append(out, it)
+ }
+ }
+ return out
+}
+
+func renderClassItems(items []classItem) string {
+ var b strings.Builder
+ for _, it := range items {
+ switch {
+ case it.posix != "":
+ b.WriteString(it.posix)
+ case it.lo == it.hi:
+ b.WriteString(classChar(it.lo))
+ default:
+ b.WriteString(classChar(it.lo))
+ b.WriteByte('-')
+ b.WriteString(classChar(it.hi))
+ }
+ }
+ return b.String()
+}
+
+// classChar escapes a byte that would otherwise be misread by the Go
+// regexp parser when emitted as a member (or range endpoint) inside a
+// character class: "]" would close the class, "^" would negate it if
+// first, "-" would start a range, "[" could open a POSIX subexpression,
+// and "\" always needs escaping to be literal.
+func classChar(c byte) string {
+ switch c {
+ case '\\', ']', '^', '-', '[':
+ return "\\" + string(c)
+ default:
+ return string(c)
+ }
+}