diff options
Diffstat (limited to 'internal/extract')
| -rw-r--r-- | internal/extract/extract.go | 6 | ||||
| -rw-r--r-- | internal/extract/plain.go | 35 |
2 files changed, 33 insertions, 8 deletions
diff --git a/internal/extract/extract.go b/internal/extract/extract.go index f8f98df..8e6c6c5 100644 --- a/internal/extract/extract.go +++ b/internal/extract/extract.go @@ -178,14 +178,14 @@ func (e *Extractor) Text(ctx context.Context, path string, size, maxRead int64) case toolExt[ext] != "": return e.legacyText(ctx, path, ext, maxRead) case markupExt[ext]: - raw, err := readDecoded(path) + raw, err := readDecoded(path, maxRead) if err != nil { return "", err } return stripMarkup(raw), nil case plainExt[ext]: - return readDecoded(path) + return readDecoded(path, maxRead) default: - return sniffText(path) + return sniffText(path, maxRead) } } diff --git a/internal/extract/plain.go b/internal/extract/plain.go index 956c3e6..8fec8c3 100644 --- a/internal/extract/plain.go +++ b/internal/extract/plain.go @@ -21,12 +21,29 @@ const sniffSize = 8192 // leading UTF-8 BOM is stripped and the rest used as is; a UTF-16 LE or BE // BOM is decoded with unicode/utf16; otherwise valid UTF-8 is used as is, // and any other invalid UTF-8 is decoded one byte per Latin-1 code point. -func readDecoded(path string) (string, error) { +func readDecoded(path string, maxRead int64) (string, error) { data, err := os.ReadFile(path) if err != nil { return "", err } - return decode(data), nil + // max-read bounds the bytes read; it must bound what they become as + // well. A file that is not valid UTF-8 decodes one byte per code point + // and so doubles, and everything downstream - normalising, folding - + // copies that again, per set of options, with several files in flight. + return bounded(decode(data), maxRead), nil +} + +// truncateAtRune is the largest cut at or below n that does not split a +// rune, so a bounded text is still valid UTF-8. +func truncateAtRune(s string, n int64) int { + i := int(n) + if i >= len(s) { + return len(s) + } + for i > 0 && !utf8.RuneStart(s[i]) { + i-- + } + return i } // sniffText decides whether an unknown extension is text, reading at most @@ -48,7 +65,7 @@ func readDecoded(path string) (string, error) { // unsupported; sample itself, used below to build the returned text, is // left untouched — the rest of the file (read after the check) supplies // the bytes trimming set aside. -func sniffText(path string) (string, error) { +func sniffText(path string, maxRead int64) (string, error) { f, err := os.Open(path) if err != nil { return "", err @@ -79,12 +96,20 @@ func sniffText(path string) (string, error) { data := append(sample, rest...) if utf16BOM { - return decode(data), nil + return bounded(decode(data), maxRead), nil } if !utf8.Valid(data) || bytes.Contains(data, []byte{0}) { return "", ErrUnsupported } - return decode(data), nil + return bounded(decode(data), maxRead), nil +} + +// bounded cuts text to maxRead at a rune boundary; 0 means unlimited. +func bounded(text string, maxRead int64) string { + if maxRead <= 0 || int64(len(text)) <= maxRead { + return text + } + return text[:truncateAtRune(text, maxRead)] } // trimIncompleteTrailingRune drops an incomplete UTF-8 sequence left |
