// SPDX-License-Identifier: GPL-3.0-or-later package engine import ( "context" "path/filepath" "sort" "strings" "sync" "time" "krino/internal/cond" "krino/internal/dup" "krino/internal/extract" "krino/internal/kwcache" "krino/internal/norm" "krino/internal/plan" "krino/internal/scan" "krino/internal/xdg" ) // matchRun holds the state shared by every file evaluated during one Match // or Explain call: the directory being matched, the full set of scanned // files (for duplicate detection) and the lazily built duplicate indexes, // one per distinct set of resolved directories a (duplicate ...) test // names. mu guards dupOnce, dupIdx and warn, the only fields any goroutine // but the one that created the matchRun ever touches. type matchRun struct { e *Engine d *Dir ctx context.Context now time.Time files []scan.File cache *kwcache.Cache // nil: no keyword cache mu sync.Mutex dupOnce map[string]*sync.Once dupIdx map[string]*dup.Index warn []string } // newMatchRun builds a matchRun over files, the set a (duplicate ...) test // with no directories of its own checks against. func newMatchRun(e *Engine, d *Dir, ctx context.Context, now time.Time, files []scan.File) *matchRun { return &matchRun{ e: e, d: d, ctx: ctx, now: now, files: files, dupOnce: make(map[string]*sync.Once), dupIdx: make(map[string]*dup.Index), } } // warnings returns the directory-level warnings collected so far (from // building duplicate indexes), in the order they were recorded. func (run *matchRun) warnings() []string { run.mu.Lock() defer run.mu.Unlock() return append([]string(nil), run.warn...) } // drainDupErrors appends every duplicate index's candidate-hashing errors // (A1: a candidate other than the file being looked up that could not be // hashed) to run.warn, once matching is done and every index has seen every // Lookup it is going to see. Candidate paths are abbreviated with // xdg.Abbrev, as every other user-visible path is. func (run *matchRun) drainDupErrors() { run.mu.Lock() defer run.mu.Unlock() for _, idx := range run.dupIdx { for _, ce := range idx.Errors() { run.warn = append(run.warn, "duplicate: "+xdg.Abbrev(ce.Path)+": "+ce.Err.Error()) } } } // dupIndex returns the shared *dup.Index for the resolved, sorted extra // directories named by key, building it exactly once across every // concurrent caller that asks for the same key. func (run *matchRun) dupIndex(key string, dirs []string) *dup.Index { run.mu.Lock() once, ok := run.dupOnce[key] if !ok { once = &sync.Once{} run.dupOnce[key] = once } run.mu.Unlock() once.Do(func() { idx, errs := dup.NewIndex(run.files, dirs) run.mu.Lock() run.dupIdx[key] = idx for _, err := range errs { run.warn = append(run.warn, "duplicate: "+err.Error()) } run.mu.Unlock() }) run.mu.Lock() idx := run.dupIdx[key] run.mu.Unlock() return idx } // facts is one file's cond.Facts. It is used by exactly one goroutine, so // its own memoised state (the keyword answers, and whether an earlier rule // matched) needs no locking of its own; only the matchRun it points at is // shared. type facts struct { run *matchRun file scan.File matched bool contentDone bool // extraction was attempted contentErr error // why it failed answers map[string]bool // by cond.KeywordKey, once extracted } var _ cond.Facts = (*facts)(nil) // newFacts builds the Facts for one scanned file. func newFacts(run *matchRun, file scan.File) *facts { return &facts{run: run, file: file} } func (f *facts) Name() string { return f.file.Name } func (f *facts) Rel() string { return f.file.Rel } func (f *facts) Size() int64 { return f.file.Size } func (f *facts) ModTime() time.Time { return f.file.ModTime } func (f *facts) Now() time.Time { return f.run.now } func (f *facts) Matched() bool { return f.matched } // ContentContains answers a content test (spec ยง6.1). A file above // max-read is never read, cached or not. Before the file has been // extracted this run, the keyword cache answers when it knows every one of // keywords for this file as it is now; otherwise the text is extracted // once, every keyword of the directory (and of this test) is answered from // it and stored in the cache, and the text itself is dropped. A failed // extraction is not cached: the next run tries again. func (f *facts) ContentContains(opt cond.Options, keywords []string) (int, error) { if max := f.run.d.Settings.MaxRead; max > 0 && f.file.Size > max { return -1, extract.ErrTooLarge } keys := make([]string, len(keywords)) for i, kw := range keywords { keys[i] = cond.KeywordKey(opt, kw) } if !f.contentDone { if id, ok := f.cacheID(); ok { if hits, ok := f.run.cache.Lookup(id, keys); ok { for i, hit := range hits { if hit { return i, nil } } return -1, nil } } f.extract(opt, keywords) } if f.contentErr != nil { return -1, f.contentErr } for i, k := range keys { if f.answers[k] { return i, nil } } return -1, nil } // extract reads the file's text and answers every keyword of the directory, // plus the asking test's (opt, keywords), from it. func (f *facts) extract(opt cond.Options, keywords []string) { f.contentDone = true text, err := f.run.e.Extract.Text(f.run.ctx, f.file.Path, f.file.Size, f.run.d.Settings.MaxRead) if err != nil { f.contentErr = err return } all := append([]cond.Keyword(nil), f.run.d.ContentKeywords...) for _, kw := range keywords { all = append(all, cond.Keyword{Opt: opt, Norm: kw}) } normed := map[cond.Options]string{} f.answers = make(map[string]bool, len(all)) for _, k := range all { t, ok := normed[k.Opt] if !ok { t = norm.Text(text, k.Opt.IgnoreCase, k.Opt.Fold) normed[k.Opt] = t } f.answers[k.Key()] = strings.Contains(t, k.Norm) } if id, ok := f.cacheID(); ok { f.run.cache.Store(id, f.answers) } } // cacheID is the file's keyword cache identity; ok is false when there is // no cache, or the platform gave the file no inode. func (f *facts) cacheID() (kwcache.ID, bool) { if f.run.cache == nil || f.file.Ino == 0 { return kwcache.ID{}, false } return fileCacheID(f.file), true } // fileCacheID is file's kwcache.ID. func fileCacheID(file scan.File) kwcache.ID { return kwcache.ID{Dev: file.Dev, Ino: file.Ino, Size: file.Size, MTime: file.ModTime.UnixNano(), Ext: strings.ToLower(filepath.Ext(file.Name))} } // Duplicate resolves dirs against the directory's root, builds (or reuses) // the shared duplicate index for that resolved, sorted set, and looks the // file up in it. func (f *facts) Duplicate(dirs []string) (string, bool, error) { root := f.run.d.Root resolved := make([]string, len(dirs)) for i, raw := range dirs { resolved[i] = plan.ResolveDir(raw, root) } sorted := append([]string(nil), resolved...) sort.Strings(sorted) key := strings.Join(sorted, "\x00") idx := f.run.dupIndex(key, sorted) orig, isDup, err := idx.Lookup(f.file.Path) if err != nil { return "", false, err } if !isDup { return "", false, nil } return displayOriginal(orig, root), true, nil } // displayOriginal reports orig relative to root when it lies inside root, // else home-abbreviated (xdg.Abbrev), as every other user-visible path is. func displayOriginal(orig, root string) string { rel, err := filepath.Rel(root, orig) if err != nil || rel == ".." || strings.HasPrefix(rel, ".."+string(filepath.Separator)) { return xdg.Abbrev(orig) } return filepath.ToSlash(rel) }