aboutsummaryrefslogtreecommitdiff
path: root/internal/extract
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-09-17 12:11:42 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-09-17 12:11:42 +0200
commitbddbd74e4a73e8e32bcf648efd1cac5655f6d0cd (patch)
tree187e6a1fb722d9ab97d2076f69f997f5d371e943 /internal/extract
parentcd7425b81f963a948f0abe7df3f9e58e190c2b78 (diff)
downloadkrino-bddbd74e4a73e8e32bcf648efd1cac5655f6d0cd.tar.gz
krino-bddbd74e4a73e8e32bcf648efd1cac5655f6d0cd.zip
comments that explain the code, not how it was written
About 340 comments cited the development process: task and plan numbers, fix waves, rulings, reviewers, and the author in the third person with a date. None of that exists outside the work itself, so to a reader it pointed at nothing. Each one now states the engineering reason it was standing in front of; where a comment was provenance and nothing else, it is gone. References to docs/design.md and docs/gui-design.md by section stay: both ship with the repository. The design documents lose their amendment diaries - CHANGELOG.md is that record - and the GUI's says plainly that the window has gone further than the document. Only comments changed. Every .go file was parsed and its code printed with comments stripped, before and after: the two hashes are identical across all 175 files.
Diffstat (limited to 'internal/extract')
-rw-r--r--internal/extract/extract.go12
-rw-r--r--internal/extract/plain.go37
-rw-r--r--internal/extract/plain_test.go14
-rw-r--r--internal/extract/zipxml_test.go15
4 files changed, 38 insertions, 40 deletions
diff --git a/internal/extract/extract.go b/internal/extract/extract.go
index c0f0262..f8f98df 100644
--- a/internal/extract/extract.go
+++ b/internal/extract/extract.go
@@ -29,7 +29,7 @@ var (
// ErrPartial wraps the error of a document read only in part (an
// archive entry that would not open or parse). Text returns the text it
// did read along with it: a keyword found there is found, but one not
- // found may be in the part that could not be read (plan 11).
+ // found may be in the part that could not be read.
ErrPartial = errors.New("part of it could not be read")
// ErrTooLarge is returned when the file is larger than the configured
// max-read; nothing is read in that case.
@@ -70,15 +70,15 @@ var plainExt = map[string]bool{
}
// zipExt is the set of Office/OpenDocument/ebook formats: a zip container
-// plus XML inside it (Task 5).
+// plus XML inside it.
var zipExt = map[string]bool{
"docx": true, "xlsx": true, "pptx": true,
"odt": true, "ods": true, "odp": true,
"epub": true,
}
-// toolExt maps an extension to the external tool it needs (Task 6). pdf is
-// handled separately, since it has its own fixed command line.
+// toolExt maps an extension to the external tool it needs. pdf is handled
+// separately, since it has its own fixed command line.
var toolExt = map[string]string{
"doc": "antiword", // falls back to catdoc
"xls": "xls2csv",
@@ -108,8 +108,8 @@ func newWithPath(path string) *Extractor {
for _, name := range toolNames {
for _, dir := range dirs {
// A relative entry would find a tool relative to the working
- // directory - a bin/pdftotext an unpacked download left behind
- // (review planapply F7) - so only absolute entries count.
+ // directory - a bin/pdftotext an unpacked download left behind -
+ // so only absolute entries count.
if dir == "" || !filepath.IsAbs(dir) {
continue
}
diff --git a/internal/extract/plain.go b/internal/extract/plain.go
index beffae1..956c3e6 100644
--- a/internal/extract/plain.go
+++ b/internal/extract/plain.go
@@ -39,13 +39,12 @@ func readDecoded(path string) (string, error) {
// sample that merely looks like UTF-8 must hold for the WHOLE file — no
// NUL byte anywhere, and no invalid UTF-8 anywhere past the sample — or
// the file is ErrUnsupported after all (a self-extracting installer has no
-// text in krino's sense, decided after the plan 10 re-check); the
-// Latin-1 fallback in decode
-// never applies to a sniffed file, only to a file whose extension already
-// names it as text. D2: when the file continues past the sample (n ==
-// sniffSize), the validity check is run against a trimmed copy with any
-// incomplete trailing rune removed, so a multi-byte rune that happens to
-// straddle byte sniffSize does not make an otherwise-valid file sniff as
+// text in krino's sense); the Latin-1 fallback in decode never applies to
+// a sniffed file, only to a file whose extension already names it as
+// text. When the file continues past the sample (n == sniffSize), the
+// validity check is run against a trimmed copy with any incomplete
+// trailing rune removed, so a multi-byte rune that happens to straddle
+// byte sniffSize does not make an otherwise-valid file sniff as
// unsupported; sample itself, used below to build the returned text, is
// left untouched — the rest of the file (read after the check) supplies
// the bytes trimming set aside.
@@ -89,13 +88,13 @@ func sniffText(path string) (string, error) {
}
// trimIncompleteTrailingRune drops an incomplete UTF-8 sequence left
-// dangling at the very end of b — D2's fix for a rune cut off exactly at
-// the sniff sample's boundary. It looks back at most utf8.UTFMax-1 bytes
-// for the start of the trailing rune; if the bytes from there to the end
-// are not a complete encoding (utf8.FullRune), that partial rune is cut,
-// since more bytes to finish it may simply not have been read yet. A
-// sample already ending cleanly (the common case, and every all-ASCII
-// sample) is returned unchanged.
+// dangling at the very end of b, so a rune cut off exactly at the sniff
+// sample's boundary is not mistaken for invalid UTF-8. It looks back at
+// most utf8.UTFMax-1 bytes for the start of the trailing rune; if the
+// bytes from there to the end are not a complete encoding (utf8.FullRune),
+// that partial rune is cut, since more bytes to finish it may simply not
+// have been read yet. A sample already ending cleanly (the common case,
+// and every all-ASCII sample) is returned unchanged.
func trimIncompleteTrailingRune(b []byte) []byte {
end := len(b)
start := end - 1
@@ -143,7 +142,7 @@ func decodeUTF16(b []byte, order binary.ByteOrder) string {
// decodeLatin1 decodes b as Latin-1: each byte is its own Unicode code
// point. Built directly as UTF-8 (at most two bytes per input byte), not
-// through a []rune of four bytes per input byte (triage 28d).
+// through a []rune of four bytes per input byte.
func decodeLatin1(b []byte) string {
var s strings.Builder
s.Grow(len(b) * 2)
@@ -200,10 +199,10 @@ func stripMarkup(s string) string {
gt := strings.IndexByte(s[j:], '>')
if gt == -1 {
- // D1: an unterminated tag (no closing '>') can no longer be
- // parsed as markup, but that is no reason to discard the rest
- // of the file - copy it through as literal text instead of
- // simply stopping the scan.
+ // An unterminated tag (no closing '>') can no longer be
+ // parsed as markup, but that is no reason to discard the
+ // rest of the file - copy it through as literal text
+ // instead of simply stopping the scan.
b.WriteString(s[i:])
break
}
diff --git a/internal/extract/plain_test.go b/internal/extract/plain_test.go
index e53ac73..9e8b747 100644
--- a/internal/extract/plain_test.go
+++ b/internal/extract/plain_test.go
@@ -156,7 +156,7 @@ func TestToolsListedInOrder(t *testing.T) {
// sample — sniffText must reject the whole file, not just decode what the
// sample alone promised (it must not fall back to Latin-1 the way a known
// text extension would). Such a file - a self-extracting installer, say -
-// has no text in krino's sense (decided after the plan 10 re-check).
+// has no text in krino's sense.
func TestSniffWholeFileMustBeValid(t *testing.T) {
e := newWithPath("")
data := append([]byte(strings.Repeat("x", 8192)), 0xFF, 0x00)
@@ -194,7 +194,7 @@ func TestToolLookupSkipsNonRegular(t *testing.T) {
}
}
-// TestUnterminatedTagKeepsRemainder: D1. An unterminated ordinary tag (no
+// TestUnterminatedTagKeepsRemainder: an unterminated ordinary tag (no
// closing '>') must not discard the rest of the file - only the malformed
// tag markup itself is unrecoverable; whatever follows it is still real
// content and must still reach the extracted text.
@@ -213,10 +213,10 @@ func TestUnterminatedTagKeepsRemainder(t *testing.T) {
}
}
-// TestSniffRuneStraddlingSampleBoundary: D2. A multi-byte rune ("ż", two
-// UTF-8 bytes) placed exactly so its lead byte is the sniff sample's last
-// byte and its continuation byte falls just past it must not make an
-// otherwise valid UTF-8 file sniff as unsupported.
+// TestSniffRuneStraddlingSampleBoundary: a multi-byte rune ("ż", two UTF-8
+// bytes) placed exactly so its lead byte is the sniff sample's last byte
+// and its continuation byte falls just past it must not make an otherwise
+// valid UTF-8 file sniff as unsupported.
func TestSniffRuneStraddlingSampleBoundary(t *testing.T) {
e := newWithPath("")
prefix := strings.Repeat("a", sniffSize-1)
@@ -232,7 +232,7 @@ func TestSniffRuneStraddlingSampleBoundary(t *testing.T) {
// TestLatin1DecodingMemory: decoding Latin-1 allocates about what the UTF-8
// result needs (at most two bytes per input byte), not a four-byte rune per
-// input byte on top of it (triage 28d).
+// input byte on top of it.
func TestLatin1DecodingMemory(t *testing.T) {
data := bytes.Repeat([]byte("Gr\xfc\xdfe "), 16000)
r := testing.Benchmark(func(b *testing.B) {
diff --git a/internal/extract/zipxml_test.go b/internal/extract/zipxml_test.go
index 30a4cf9..5b7ba54 100644
--- a/internal/extract/zipxml_test.go
+++ b/internal/extract/zipxml_test.go
@@ -128,10 +128,9 @@ func TestOdfEmbeddedObjectIncluded(t *testing.T) {
// The footer's mismatched end tag (</w:xyz> where </w:ftr> was open)
// genuinely fails to parse under Strict=false — unlike a merely missing
// end tag, which the decoder synthesises and swallows — but only after
-// "broken" and the newline for </w:p> have already been written; the
-// XML decoder itself confirms this token by token (see fix round 2's
-// report for the trace): CharData "broken", EndElement p, then the
-// error "unexpected end element </xyz>".
+// "broken" and the newline for </w:p> have already been written; the XML
+// decoder itself confirms this token by token: CharData "broken",
+// EndElement p, then the error "unexpected end element </xyz>".
func TestZipLenientOnMalformedEntry(t *testing.T) {
e := newWithPath("")
got, err := text(t, e, zipFile(t, "partial.docx", map[string]string{
@@ -139,7 +138,7 @@ func TestZipLenientOnMalformedEntry(t *testing.T) {
"word/footer1.xml": `<w:ftr xmlns:w="w"><w:p><w:r><w:t>broken</w:t></w:r></w:p></w:xyz>`,
}), 0)
// Partly read: the text is kept, and ErrPartial says some of it is
- // missing, so a keyword not found in it is unknown (plan 11).
+ // missing, so a keyword not found in it is unknown.
if !errors.Is(err, ErrPartial) {
t.Fatalf("good entry alongside a malformed one: err %v, want ErrPartial", err)
}
@@ -193,9 +192,9 @@ func TestZipNoTextAndFailingSiblingErrors(t *testing.T) {
}
}
-// TestMaxReadCapsZipOutput: B1. A directory's max-read, when smaller than
-// the fixed zipBudget default, caps a single archive's extracted text on
-// its own — zipBudget is left at its default, so only budget(zipBudget,
+// TestMaxReadCapsZipOutput: a directory's max-read, when smaller than the
+// fixed zipBudget default, caps a single archive's extracted text on its
+// own — zipBudget is left at its default, so only budget(zipBudget,
// maxRead) picking the smaller maxRead explains the result.
func TestMaxReadCapsZipOutput(t *testing.T) {
e := newWithPath("")