From 1506c7dcd6032c04c1df6f5785b6dd3dfe4511cc Mon Sep 17 00:00:00 2001 From: Lukasz Kasprzak Date: Mon, 14 Sep 2026 10:56:21 +0200 Subject: krino: duplicates are found, never deleted A rule combining (duplicate) with a delete action is refused at load, and a file that is a duplicate under any duplicate scope its directory uses gets no delete step from any rule: the plan shows it skipped and the chain continues. A failed duplicate check blocks the delete too. Tests cover (matched), another rule's own condition, a test evaluation skipped, two scopes and a failed check, each checking every copy is still on disk. --- internal/engine/engine.go | 50 +++++++++ internal/engine/engine_test.go | 44 ++++++++ internal/engine/match.go | 42 ++++++++ internal/engine/match_test.go | 2 +- internal/engine/nodelete_test.go | 212 +++++++++++++++++++++++++++++++++++++++ internal/engine/plan.go | 2 +- internal/plan/chain.go | 8 ++ internal/plan/chain_test.go | 28 ++++++ 8 files changed, 386 insertions(+), 2 deletions(-) create mode 100644 internal/engine/nodelete_test.go (limited to 'internal') diff --git a/internal/engine/engine.go b/internal/engine/engine.go index ea61e8c..b5f2106 100644 --- a/internal/engine/engine.go +++ b/internal/engine/engine.go @@ -8,6 +8,7 @@ package engine import ( "fmt" "os" + "strings" "time" "krino/internal/cond" @@ -44,6 +45,12 @@ type Dir struct { // other variant will ever be asked for; with more than one, both must // stay memoised, as before. ContentVariants []cond.Options + + // DupScopes is every distinct directory list the duplicate tests of + // Rules use, in first-seen order, the plain (duplicate) as an empty + // list. Spec §5.5 rule 2 looks a file up under each of them before any + // rule may delete it. + DupScopes [][]string } // Rule is one directory's rule, with its condition compiled. @@ -90,9 +97,14 @@ func Load(mainFile string, names ...string) (*Engine, []*config.Diag) { errs = append(errs, diag) continue } + if diag := checkDuplicateDelete(d.File, r, c); diag != nil { + errs = append(errs, diag) + continue + } dir.Rules = append(dir.Rules, &Rule{Name: r.Name, Conf: r, Settings: rs, Cond: c}) } dir.ContentVariants = contentVariants(dir.Rules) + dir.DupScopes = dupScopes(dir.Rules) dirs = append(dirs, dir) } @@ -132,6 +144,23 @@ func checkCaptures(file string, r *config.Rule, c *cond.Cond) *config.Diag { return nil } +// checkDuplicateDelete refuses a rule that combines a duplicate test with a +// delete action (spec §4.5, §5.5): duplicates are found, never deleted. +// Cond.DupDirs records every duplicate test compiled, inside or and not +// too, so a test anywhere in the condition counts. It reports only the +// first delete action, so one config mistake yields one diagnostic. +func checkDuplicateDelete(file string, r *config.Rule, c *cond.Cond) *config.Diag { + if len(c.DupDirs) == 0 { + return nil + } + for _, a := range r.Actions { + if a.Kind == config.Delete || a.Kind == config.DeletePermanent { + return &config.Diag{File: file, Pos: a.Pos, Msg: fmt.Sprintf("rule %q: (duplicate) cannot be combined with (%s): duplicates are never deleted, move them aside instead", r.Name, a.Kind)} + } + } + return nil +} + // captureGroups renders a capture-group count with correct singular/plural. func captureGroups(n int) string { if n == 1 { @@ -177,6 +206,27 @@ func contentVariants(rules []*Rule) []cond.Options { return out } +// dupScopes returns the distinct Cond.DupDirs lists of rules, in first-seen +// order. Lists are compared as written, with their length in the key so +// (duplicate) and (duplicate "") stay apart; two spellings of one directory +// stay two entries, which costs a second lookup but never a wrong answer, +// since facts.Duplicate resolves and shares the index itself. +func dupScopes(rules []*Rule) [][]string { + var out [][]string + seen := map[string]bool{} + for _, r := range rules { + for _, dirs := range r.Cond.DupDirs { + key := fmt.Sprintf("%d\x00%s", len(dirs), strings.Join(dirs, "\x00")) + if seen[key] { + continue + } + seen[key] = true + out = append(out, dirs) + } + } + return out +} + // Report is what Check reports: the files involved and each directory's // state. type Report struct { diff --git a/internal/engine/engine_test.go b/internal/engine/engine_test.go index 8dcfcb3..18b35e7 100644 --- a/internal/engine/engine_test.go +++ b/internal/engine/engine_test.go @@ -213,3 +213,47 @@ func TestContentVariantsComputedAtLoad(t *testing.T) { t.Fatalf("ContentVariants = %+v, want %+v", got, want) } } + +// TestLoadRefusesDuplicateWithDelete is spec §4.5: a (duplicate) test +// anywhere in a rule's condition, and a delete action in the same rule, is +// a load error, however the test is nested. +func TestLoadRefusesDuplicateWithDelete(t *testing.T) { + tests := []struct{ rule, want string }{ + {`(rule "a" (when (duplicate)) (delete))`, + `rule "a": (duplicate) cannot be combined with (delete): duplicates are never deleted, move them aside instead`}, + {`(rule "a" (when (duplicate "Archive")) (delete permanent))`, + `rule "a": (duplicate) cannot be combined with (delete permanent)`}, + {`(rule "a" (when (or (type pdf) (duplicate))) (move "Keep") (delete))`, + `rule "a": (duplicate) cannot be combined with (delete)`}, + {`(rule "a" (when (not (duplicate))) (delete))`, + `rule "a": (duplicate) cannot be combined with (delete)`}, + } + for _, tt := range tests { + h := sandbox(t) + main := writeConfig(t, h, `(include "dl")`, map[string]string{"dl": `(path "/tmp") +` + tt.rule}) + _, errs := Load(main, "dl") + joined := "" + for _, d := range errs { + joined += d.Error() + "\n" + } + if len(errs) != 1 || !strings.Contains(joined, tt.want) { + t.Errorf("rule %s: errs %v, want %q", tt.rule, errs, tt.want) + } + } +} + +// TestLoadAcceptsDuplicateWithMove: the way §5.5 recommends dealing with +// duplicates loads clean, and so does a delete rule with no duplicate test. +func TestLoadAcceptsDuplicateWithMove(t *testing.T) { + for _, rule := range []string{ + `(rule "dupes" (when (duplicate "Archive")) (move "~/.dupes/") (stop))`, + `(rule "old" (when (type iso) (age > 90d)) (delete))`, + } { + h := sandbox(t) + main := writeConfig(t, h, `(include "dl")`, map[string]string{"dl": "(path \"/tmp\")\n" + rule}) + if _, errs := Load(main, "dl"); len(errs) != 0 { + t.Errorf("rule %s: errs %v, want none", rule, errs) + } + } +} diff --git a/internal/engine/match.go b/internal/engine/match.go index e68a6e8..ffee7f0 100644 --- a/internal/engine/match.go +++ b/internal/engine/match.go @@ -33,6 +33,12 @@ type FileMatch struct { File scan.File Rules []RuleMatch // matching rules in order, ending at the first with (stop) Warnings []string // ": ", e.g. "acme: content unreadable: needs pdftotext, not installed" + + // NoDelete is non-empty when no delete step may run for this file (spec + // §5.5 rule 2), and says why: the file is a duplicate under a scope its + // directory's rules use, or that check failed. Only set for a file some + // matching rule would delete. + NoDelete string } // Result is everything Match found in one directory. @@ -127,9 +133,45 @@ func evalFile(run *matchRun, file scan.File) FileMatch { break } } + if len(run.d.DupScopes) > 0 && deletes(fm.Rules) { + fm.NoDelete = noDelete(f, run.d.DupScopes) + } return fm } +// NeverDeleted is the reason a duplicate's delete step is skipped (spec +// §5.5). +const NeverDeleted = "a duplicate is never deleted" + +// deletes reports whether any of rules has a delete action. +func deletes(rules []RuleMatch) bool { + for _, rm := range rules { + for _, a := range rm.Rule.Conf.Actions { + if a.Kind == config.Delete || a.Kind == config.DeletePermanent { + return true + } + } + } + return false +} + +// noDelete looks f up under every scope, whether or not evaluating the +// rules reached that test (spec §5.5 rule 2), and returns why f must not be +// deleted, or "" when it may be. A failed lookup blocks the delete too: +// krino cannot show the file is not a duplicate, so it keeps it. +func noDelete(f *facts, scopes [][]string) string { + for _, dirs := range scopes { + _, dup, err := f.Duplicate(dirs) + if err != nil { + return "duplicate check failed, so not deleted: " + err.Error() + } + if dup { + return NeverDeleted + } + } + return "" +} + // RuleTrace is one rule's outcome in an Explain call. type RuleTrace struct { Rule *Rule diff --git a/internal/engine/match_test.go b/internal/engine/match_test.go index 1356074..477e06f 100644 --- a/internal/engine/match_test.go +++ b/internal/engine/match_test.go @@ -18,7 +18,7 @@ const dlConf = ` (recursive yes) (min-age 0s) (ignore "*.part") -(rule "dups" (when (duplicate)) (delete) (stop)) +(rule "dups" (when (duplicate)) (move "Dupes") (stop)) (rule "acme" (when (type document) (content "acme ltd")) (move "Work/Acme") (stop)) (rule "images" (when (type image)) (move "Pictures")) (rule "rest" (when (not (matched)) (type text)) (move "Other")) diff --git a/internal/engine/nodelete_test.go b/internal/engine/nodelete_test.go new file mode 100644 index 0000000..80d9ed9 --- /dev/null +++ b/internal/engine/nodelete_test.go @@ -0,0 +1,212 @@ +// SPDX-License-Identifier: GPL-3.0-or-later + +package engine + +import ( + "context" + "io/fs" + "os" + "path/filepath" + "sort" + "strings" + "testing" + "time" + + "krino/internal/journal" + "krino/internal/plan" +) + +// sameBytes is the content every file in these tests shares. +const sameBytes = "%PDF identical bytes" + +// dlTree creates ~/dl in a sandbox. Each file holds body and is modified +// the given number of hours after a fixed old time, so the smallest number +// is the oldest file. PATH is emptied so no extraction tool runs. +func dlTree(t *testing.T, files map[string]int, body string) (home, dl string) { + t.Helper() + home = sandbox(t) + t.Setenv("PATH", t.TempDir()) + dl = filepath.Join(home, "dl") + old := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + for rel, hours := range files { + p := filepath.Join(dl, rel) + if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(p, []byte(body), 0o644); err != nil { + t.Fatal(err) + } + mt := old.Add(time.Duration(hours) * time.Hour) + if err := os.Chtimes(p, mt, mt); err != nil { + t.Fatal(err) + } + } + return home, dl +} + +// planAndApply configures ~/dl (recursive, min-age 0s) with rules, plans +// it, approves and applies every chain, and returns the plan. +func planAndApply(t *testing.T, home, rules string) *DirPlan { + t.Helper() + conf := "(path \"~/dl\")\n(recursive yes)\n(min-age 0s)\n" + rules + main := writeConfig(t, home, `(include "dl")`, map[string]string{"dl": conf}) + e, errs := Load(main) + if len(errs) > 0 { + t.Fatal(errs) + } + dp, err := e.Plan(context.Background(), e.Dirs[0], plan.NewClaims()) + if err != nil { + t.Fatal(err) + } + approved := map[string]bool{} + for _, c := range dp.Chains { + approved[c.File.Rel] = true + } + j, err := journal.Open(filepath.Join(home, ".local", "state", "krino", "krino.log")) + if err != nil { + t.Fatal(err) + } + defer j.Close() + res, err := e.Apply(context.Background(), dp, approved, j, journal.NewRunID(time.Now())) + if err != nil { + t.Fatal(err) + } + if res.Failed != 0 { + t.Fatalf("%d files failed: %+v", res.Failed, res) + } + return dp +} + +// assertCopies checks that exactly the files want (paths relative to dl, +// in any order) hold body after the run. +func assertCopies(t *testing.T, dl, body string, want []string) { + t.Helper() + var got []string + err := filepath.WalkDir(dl, func(p string, d fs.DirEntry, err error) error { + if err != nil || !d.Type().IsRegular() { + return err + } + if b, err := os.ReadFile(p); err == nil && string(b) == body { + rel, _ := filepath.Rel(dl, p) + got = append(got, filepath.ToSlash(rel)) + } + return nil + }) + if err != nil { + t.Fatal(err) + } + sort.Strings(got) + w := append([]string(nil), want...) + sort.Strings(w) + if strings.Join(got, "\n") != strings.Join(w, "\n") { + t.Errorf("copies on disk = %q, want %q", got, w) + } +} + +// assertSkip checks that rel's chain has a step from rule skipped with a +// reason starting with want. +func assertSkip(t *testing.T, dp *DirPlan, rel, rule, want string) { + t.Helper() + for _, c := range dp.Chains { + if c.File.Rel != rel { + continue + } + for _, s := range c.Steps { + if s.Rule == rule && strings.HasPrefix(s.Skip, want) { + return + } + } + t.Errorf("%s: no step of rule %q skipped with %q; steps %+v", rel, rule, want, c.Steps) + return + } + t.Errorf("%s: no chain in the plan", rel) +} + +// TestNoDeleteThroughMatched: a later rule deleting through (matched) +// cannot delete what an earlier duplicate rule found. +func TestNoDeleteThroughMatched(t *testing.T) { + h, dl := dlTree(t, map[string]int{"a.pdf": 0, "b.pdf": 1}, sameBytes) + dp := planAndApply(t, h, ` +(rule "dupes" (when (duplicate)) (move "Dupes")) +(rule "cleanup" (when (matched)) (delete permanent)) +`) + assertCopies(t, dl, sameBytes, []string{"a.pdf", "Dupes/b.pdf"}) + assertSkip(t, dp, "b.pdf", "cleanup", NeverDeleted) +} + +// TestNoDeleteThroughAnotherRulesCondition: a rule with no duplicate test +// of its own deletes the original, but not the duplicate. +func TestNoDeleteThroughAnotherRulesCondition(t *testing.T) { + h, dl := dlTree(t, map[string]int{"a.pdf": 0, "b.pdf": 1}, sameBytes) + dp := planAndApply(t, h, ` +(rule "pdfs" (when (type pdf)) (delete permanent)) +(rule "dupes" (when (duplicate)) (move "Dupes")) +`) + assertCopies(t, dl, sameBytes, []string{"Dupes/b.pdf"}) + assertSkip(t, dp, "b.pdf", "pdfs", NeverDeleted) +} + +// TestNoDeleteWhenEvaluationSkippedTheDuplicateTest: "dupes" tests size +// first, which is false, so its duplicate test is never evaluated; b.pdf +// is still a duplicate under that scope, so "pdfs" cannot delete it. +func TestNoDeleteWhenEvaluationSkippedTheDuplicateTest(t *testing.T) { + h, dl := dlTree(t, map[string]int{"a.pdf": 0, "b.pdf": 1}, sameBytes) + dp := planAndApply(t, h, ` +(rule "dupes" (when (size > 1G) (duplicate)) (move "Dupes")) +(rule "pdfs" (when (type pdf)) (delete permanent)) +`) + assertCopies(t, dl, sameBytes, []string{"b.pdf"}) + assertSkip(t, dp, "b.pdf", "pdfs", NeverDeleted) +} + +// TestNoDeleteWithTwoScopesWhoseOriginalsDiffer: under "Archive" the +// original is Archive/x.pdf, under the plain scope it is the older +// x-copy.pdf, so each file is a duplicate somewhere. Neither is deleted; +// both are moved aside. +func TestNoDeleteWithTwoScopesWhoseOriginalsDiffer(t *testing.T) { + h, dl := dlTree(t, map[string]int{"Archive/x.pdf": 1, "x-copy.pdf": 0}, sameBytes) + dp := planAndApply(t, h, ` +(rule "pdfs" (when (type pdf)) (delete permanent)) +(rule "archive-dupes" (when (duplicate "Archive")) (move "Dupes") (stop)) +(rule "local-dupes" (when (duplicate)) (move "Dupes") (stop)) +`) + assertCopies(t, dl, sameBytes, []string{"Dupes/x.pdf", "Dupes/x-copy.pdf"}) + assertSkip(t, dp, "Archive/x.pdf", "pdfs", NeverDeleted) + assertSkip(t, dp, "x-copy.pdf", "pdfs", NeverDeleted) +} + +// TestDeleteWithoutDuplicateTestsStillDeletes: the guarantee applies only +// where a directory's rules use (duplicate); a plain delete rule still does +// what it says. +func TestDeleteWithoutDuplicateTestsStillDeletes(t *testing.T) { + h, dl := dlTree(t, map[string]int{"a.pdf": 0, "b.pdf": 1}, sameBytes) + planAndApply(t, h, `(rule "pdfs" (when (type pdf)) (delete permanent))`) + assertCopies(t, dl, sameBytes, nil) +} + +// TestNoDeleteWhenTheDuplicateCheckFails: b.pdf cannot be read, so krino +// cannot show it is not a duplicate; its delete is skipped rather than +// guessed. a.pdf, whose only candidate could not be hashed, is not a +// duplicate and is deleted. +func TestNoDeleteWhenTheDuplicateCheckFails(t *testing.T) { + if os.Geteuid() == 0 { + t.Skip("root reads files regardless of their mode") + } + h, dl := dlTree(t, map[string]int{"a.pdf": 0, "b.pdf": 1}, sameBytes) + b := filepath.Join(dl, "b.pdf") + if err := os.Chmod(b, 0); err != nil { + t.Fatal(err) + } + t.Cleanup(func() { os.Chmod(b, 0o644) }) + dp := planAndApply(t, h, ` +(rule "dupes" (when (size > 1G) (duplicate)) (move "Dupes")) +(rule "pdfs" (when (type pdf)) (delete)) +`) + assertSkip(t, dp, "b.pdf", "pdfs", "duplicate check failed, so not deleted: ") + if _, err := os.Lstat(b); err != nil { + t.Errorf("b.pdf is gone: %v", err) + } + if _, err := os.Lstat(filepath.Join(dl, "a.pdf")); !os.IsNotExist(err) { + t.Errorf("a.pdf was not deleted: %v", err) + } +} diff --git a/internal/engine/plan.go b/internal/engine/plan.go index 979bec1..d9ffeb3 100644 --- a/internal/engine/plan.go +++ b/internal/engine/plan.go @@ -52,7 +52,7 @@ func (e *Engine) Plan(ctx context.Context, d *Dir, claims *plan.Claims) (*DirPla Reasons: rm.Reasons, } } - inputs[i] = plan.Input{File: fm.File, Rules: rules} + inputs[i] = plan.Input{File: fm.File, Rules: rules, NoDelete: fm.NoDelete} } chains := plan.Build(d.Root, inputs, e.Now(), plan.OS{}, claims) diff --git a/internal/plan/chain.go b/internal/plan/chain.go index bfa2484..e504504 100644 --- a/internal/plan/chain.go +++ b/internal/plan/chain.go @@ -32,6 +32,10 @@ func NewClaims() *Claims { type Input struct { File scan.File Rules []RuleMatch + // NoDelete, when non-empty, skips every delete step of this file with + // this text as the step's Skip, without ending the chain (spec §5.5 + // rule 2, §7.1). + NoDelete string } // Build turns each file's matching rules into a chain. root is the @@ -166,6 +170,10 @@ func buildOne(root string, in Input, now time.Time, d Disk, claim claimed) Chain } case config.Delete, config.DeletePermanent: + if in.NoDelete != "" { + step.Skip = in.NoDelete + break + } deletedBy = rule.Name } diff --git a/internal/plan/chain_test.go b/internal/plan/chain_test.go index b6e3dfe..28c208f 100644 --- a/internal/plan/chain_test.go +++ b/internal/plan/chain_test.go @@ -141,3 +141,31 @@ func TestBuildKeepsSteplessChains(t *testing.T) { } } } + +// TestBuildNoDeleteSkipsDeletesAndContinues is spec §5.5 rule 2 and §7.1: +// with Input.NoDelete set, every delete step is skipped with that reason, +// and the chain carries on from the file's current path instead of ending. +func TestBuildNoDeleteSkipsDeletesAndContinues(t *testing.T) { + in := []Input{{ + File: file("/r", "b.pdf"), + NoDelete: "a duplicate is never deleted", + Rules: []RuleMatch{ + {Name: "old", Actions: []config.Action{act(config.DeletePermanent, "")}}, + {Name: "dupes", Actions: []config.Action{act(config.Move, "Dupes")}}, + {Name: "cleanup", Actions: []config.Action{act(config.Delete, "")}}, + }, + }} + steps := Build("/r", in, time.Now(), NoDisk{}, NewClaims())[0].Steps + if len(steps) != 3 { + t.Fatalf("steps = %+v", steps) + } + if steps[0].Kind != DeletePermanent || steps[0].Skip != "a duplicate is never deleted" { + t.Errorf("step 0 = %+v, want a skipped permanent delete", steps[0]) + } + if steps[1].Kind != Move || steps[1].Skip != "" || steps[1].Dst != "/r/Dupes/b.pdf" { + t.Errorf("step 1 = %+v, want the move to run", steps[1]) + } + if steps[2].Kind != Trash || steps[2].Skip != "a duplicate is never deleted" || steps[2].Src != "/r/Dupes/b.pdf" { + t.Errorf("step 2 = %+v, want a skipped trash from the moved path", steps[2]) + } +} -- cgit v1.3