aboutsummaryrefslogtreecommitdiff
path: root/internal/extract
diff options
context:
space:
mode:
Diffstat (limited to 'internal/extract')
-rw-r--r--internal/extract/extract.go6
-rw-r--r--internal/extract/plain.go35
2 files changed, 33 insertions, 8 deletions
diff --git a/internal/extract/extract.go b/internal/extract/extract.go
index f8f98df..8e6c6c5 100644
--- a/internal/extract/extract.go
+++ b/internal/extract/extract.go
@@ -178,14 +178,14 @@ func (e *Extractor) Text(ctx context.Context, path string, size, maxRead int64)
case toolExt[ext] != "":
return e.legacyText(ctx, path, ext, maxRead)
case markupExt[ext]:
- raw, err := readDecoded(path)
+ raw, err := readDecoded(path, maxRead)
if err != nil {
return "", err
}
return stripMarkup(raw), nil
case plainExt[ext]:
- return readDecoded(path)
+ return readDecoded(path, maxRead)
default:
- return sniffText(path)
+ return sniffText(path, maxRead)
}
}
diff --git a/internal/extract/plain.go b/internal/extract/plain.go
index 956c3e6..8fec8c3 100644
--- a/internal/extract/plain.go
+++ b/internal/extract/plain.go
@@ -21,12 +21,29 @@ const sniffSize = 8192
// leading UTF-8 BOM is stripped and the rest used as is; a UTF-16 LE or BE
// BOM is decoded with unicode/utf16; otherwise valid UTF-8 is used as is,
// and any other invalid UTF-8 is decoded one byte per Latin-1 code point.
-func readDecoded(path string) (string, error) {
+func readDecoded(path string, maxRead int64) (string, error) {
data, err := os.ReadFile(path)
if err != nil {
return "", err
}
- return decode(data), nil
+ // max-read bounds the bytes read; it must bound what they become as
+ // well. A file that is not valid UTF-8 decodes one byte per code point
+ // and so doubles, and everything downstream - normalising, folding -
+ // copies that again, per set of options, with several files in flight.
+ return bounded(decode(data), maxRead), nil
+}
+
+// truncateAtRune is the largest cut at or below n that does not split a
+// rune, so a bounded text is still valid UTF-8.
+func truncateAtRune(s string, n int64) int {
+ i := int(n)
+ if i >= len(s) {
+ return len(s)
+ }
+ for i > 0 && !utf8.RuneStart(s[i]) {
+ i--
+ }
+ return i
}
// sniffText decides whether an unknown extension is text, reading at most
@@ -48,7 +65,7 @@ func readDecoded(path string) (string, error) {
// unsupported; sample itself, used below to build the returned text, is
// left untouched — the rest of the file (read after the check) supplies
// the bytes trimming set aside.
-func sniffText(path string) (string, error) {
+func sniffText(path string, maxRead int64) (string, error) {
f, err := os.Open(path)
if err != nil {
return "", err
@@ -79,12 +96,20 @@ func sniffText(path string) (string, error) {
data := append(sample, rest...)
if utf16BOM {
- return decode(data), nil
+ return bounded(decode(data), maxRead), nil
}
if !utf8.Valid(data) || bytes.Contains(data, []byte{0}) {
return "", ErrUnsupported
}
- return decode(data), nil
+ return bounded(decode(data), maxRead), nil
+}
+
+// bounded cuts text to maxRead at a rune boundary; 0 means unlimited.
+func bounded(text string, maxRead int64) string {
+ if maxRead <= 0 || int64(len(text)) <= maxRead {
+ return text
+ }
+ return text[:truncateAtRune(text, maxRead)]
}
// trimIncompleteTrailingRune drops an incomplete UTF-8 sequence left