aboutsummaryrefslogtreecommitdiff
path: root/internal/extract/plain.go
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-09-17 14:07:37 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-09-17 14:07:37 +0200
commit4c6fadfab5434317357ad0272ace7927d2945942 (patch)
treed25c3770f0c25fa07034abec2e9bbad214e12eee /internal/extract/plain.go
parent33d771438f1da363af7c9ff2d5f4c368c326e84d (diff)
downloadkrino-4c6fadfab5434317357ad0272ace7927d2945942.tar.gz
krino-4c6fadfab5434317357ad0272ace7927d2945942.zip
max-read bounds what a file becomes, not only what is read
max-read gates on file size before reading, then the text was read whole and decoded: a file that is not valid UTF-8 decodes one byte per code point and doubles, and normalising and folding copy that again per set of options, with GOMAXPROCS files in flight. Twelve 40 MB files reached 3.3 GB - enough to put a laptop into the OOM killer, with no attacker involved, just a few big .log or .csv files. Two bounds. The decoded text is cut to max-read at a rune boundary, so the ceiling means what a reader takes it to mean. And extraction of files at or above 4 MiB is rationed to two at a time, since holding several large texts at once is what multiplies the ceiling; smaller files, which is nearly all of them, are untouched. twelve 40 MB files: peak RSS 3294 MB -> 728 MB, wall 27s -> 46s two thousand small files: 0.05s both ways The wall-clock cost falls entirely on large files needing extraction, and buys a program that finishes instead of being killed.
Diffstat (limited to 'internal/extract/plain.go')
-rw-r--r--internal/extract/plain.go35
1 files changed, 30 insertions, 5 deletions
diff --git a/internal/extract/plain.go b/internal/extract/plain.go
index 956c3e6..8fec8c3 100644
--- a/internal/extract/plain.go
+++ b/internal/extract/plain.go
@@ -21,12 +21,29 @@ const sniffSize = 8192
// leading UTF-8 BOM is stripped and the rest used as is; a UTF-16 LE or BE
// BOM is decoded with unicode/utf16; otherwise valid UTF-8 is used as is,
// and any other invalid UTF-8 is decoded one byte per Latin-1 code point.
-func readDecoded(path string) (string, error) {
+func readDecoded(path string, maxRead int64) (string, error) {
data, err := os.ReadFile(path)
if err != nil {
return "", err
}
- return decode(data), nil
+ // max-read bounds the bytes read; it must bound what they become as
+ // well. A file that is not valid UTF-8 decodes one byte per code point
+ // and so doubles, and everything downstream - normalising, folding -
+ // copies that again, per set of options, with several files in flight.
+ return bounded(decode(data), maxRead), nil
+}
+
+// truncateAtRune is the largest cut at or below n that does not split a
+// rune, so a bounded text is still valid UTF-8.
+func truncateAtRune(s string, n int64) int {
+ i := int(n)
+ if i >= len(s) {
+ return len(s)
+ }
+ for i > 0 && !utf8.RuneStart(s[i]) {
+ i--
+ }
+ return i
}
// sniffText decides whether an unknown extension is text, reading at most
@@ -48,7 +65,7 @@ func readDecoded(path string) (string, error) {
// unsupported; sample itself, used below to build the returned text, is
// left untouched — the rest of the file (read after the check) supplies
// the bytes trimming set aside.
-func sniffText(path string) (string, error) {
+func sniffText(path string, maxRead int64) (string, error) {
f, err := os.Open(path)
if err != nil {
return "", err
@@ -79,12 +96,20 @@ func sniffText(path string) (string, error) {
data := append(sample, rest...)
if utf16BOM {
- return decode(data), nil
+ return bounded(decode(data), maxRead), nil
}
if !utf8.Valid(data) || bytes.Contains(data, []byte{0}) {
return "", ErrUnsupported
}
- return decode(data), nil
+ return bounded(decode(data), maxRead), nil
+}
+
+// bounded cuts text to maxRead at a rune boundary; 0 means unlimited.
+func bounded(text string, maxRead int64) string {
+ if maxRead <= 0 || int64(len(text)) <= maxRead {
+ return text
+ }
+ return text[:truncateAtRune(text, maxRead)]
}
// trimIncompleteTrailingRune drops an incomplete UTF-8 sequence left