aboutsummaryrefslogtreecommitdiff
path: root/internal/engine/facts.go
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-09-17 14:07:37 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-09-17 14:07:37 +0200
commit4c6fadfab5434317357ad0272ace7927d2945942 (patch)
treed25c3770f0c25fa07034abec2e9bbad214e12eee /internal/engine/facts.go
parent33d771438f1da363af7c9ff2d5f4c368c326e84d (diff)
downloadkrino-4c6fadfab5434317357ad0272ace7927d2945942.tar.gz
krino-4c6fadfab5434317357ad0272ace7927d2945942.zip
max-read bounds what a file becomes, not only what is read
max-read gates on file size before reading, then the text was read whole and decoded: a file that is not valid UTF-8 decodes one byte per code point and doubles, and normalising and folding copy that again per set of options, with GOMAXPROCS files in flight. Twelve 40 MB files reached 3.3 GB - enough to put a laptop into the OOM killer, with no attacker involved, just a few big .log or .csv files. Two bounds. The decoded text is cut to max-read at a rune boundary, so the ceiling means what a reader takes it to mean. And extraction of files at or above 4 MiB is rationed to two at a time, since holding several large texts at once is what multiplies the ceiling; smaller files, which is nearly all of them, are untouched. twelve 40 MB files: peak RSS 3294 MB -> 728 MB, wall 27s -> 46s two thousand small files: 0.05s both ways The wall-clock cost falls entirely on large files needing extraction, and buys a program that finishes instead of being killed.
Diffstat (limited to 'internal/engine/facts.go')
-rw-r--r--internal/engine/facts.go26
1 files changed, 26 insertions, 0 deletions
diff --git a/internal/engine/facts.go b/internal/engine/facts.go
index 94ac3f4..6540fdd 100644
--- a/internal/engine/facts.go
+++ b/internal/engine/facts.go
@@ -37,12 +37,28 @@ type matchRun struct {
files []scan.File
cache *kwcache.Cache // nil: no keyword cache
+ // bigText limits how many large files have their text in memory at
+ // once. Extracting is otherwise GOMAXPROCS-wide, and a large file's
+ // text costs several times its own size: it is read whole, a file that
+ // is not valid UTF-8 doubles as it decodes, and each distinct set of
+ // case/fold options makes another normalised copy. Sixteen 49 MB text
+ // files reached 2.7 GB. Small files are unaffected, which is nearly
+ // every file.
+ bigText chan struct{}
+
mu sync.Mutex
dupOnce map[string]*sync.Once
dupIdx map[string]*dup.Index
warn []string
}
+// bigTextSize is the file size above which extraction is rationed, and
+// bigTextAtOnce is how many such files may be in flight together.
+const (
+ bigTextSize = 4 << 20 // 4 MiB
+ bigTextAtOnce = 2
+)
+
// newMatchRun builds a matchRun over files, the set a (duplicate ...) test
// with no directories of its own checks against.
func newMatchRun(e *Engine, d *Dir, ctx context.Context, now time.Time, files []scan.File) *matchRun {
@@ -54,6 +70,7 @@ func newMatchRun(e *Engine, d *Dir, ctx context.Context, now time.Time, files []
files: files,
dupOnce: make(map[string]*sync.Once),
dupIdx: make(map[string]*dup.Index),
+ bigText: make(chan struct{}, bigTextAtOnce),
}
}
@@ -225,6 +242,15 @@ func (f *facts) ContentContains(opt cond.Options, keywords []string) (int, error
// plus the asking test's (opt, keywords), from it.
func (f *facts) extract(opt cond.Options, keywords []string) {
f.contentDone = true
+ if f.file.Size >= bigTextSize && f.run.bigText != nil {
+ // Wait for a slot rather than hold several large files' text at
+ // once. Cancellation is still noticed: Text takes the same ctx.
+ select {
+ case f.run.bigText <- struct{}{}:
+ defer func() { <-f.run.bigText }()
+ case <-f.run.ctx.Done():
+ }
+ }
text, err := f.run.e.Extract.Text(f.run.ctx, f.file.Path, f.file.Size, f.run.d.Settings.MaxRead)
switch {
case errors.Is(err, extract.ErrPartial):