summaryrefslogtreecommitdiff
path: root/internal/bible/validate.go
blob: 27ec519be054efa47c3c5c3f7f593e781a05ea35 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
package bible

import (
	"fmt"
	"sort"
)

// CorpusReport is the outcome of validating one corpus's sidecar + text (see
// CheckCorpus). Errors mean the corpus is unusable/malformed and fail
// --corpus-check (exit 1); Warnings flag coverage gaps against the reference
// Vulgate ("vul") and never fail the check -- a corpus may legitimately be
// incomplete (the built-in wuj is) and still be a valid drop-in.
type CorpusReport struct {
	Code     string
	Errors   []string
	Warnings []string
}

// OK reports whether the corpus is usable: no errors. Warnings never affect it.
func (r CorpusReport) OK() bool { return len(r.Errors) == 0 }

var validPsalmSystems = map[string]bool{"vulgate": true, "hebrew": true, "drb": true}

// CheckCorpus validates code's sidecar and text and reports coverage gaps vs
// "vul" as warnings.
//
// Sidecar: lang, name and psalm_system are required (psalm_system must be
// vulgate/hebrew/drb); sigla and autoselect are optional -- their absence, or
// autoselect=false, is never flagged (the built-in wuj sets autoselect=false
// and must pass clean).
//
// Text: every Book column value must be one of CanonicalBooks' 73 keys
// (error). A chapter missing entirely versus vul, a verse-number gap within a
// chapter, or a repeated verse number within a chapter, is a warning only --
// real source texts legitimately do this (e.g. the LXX's lettered doublet
// verses in 3 Kingdoms, or a scanned translation's occasional merged verse),
// and it must never turn a corpus that otherwise loads fine into a failure.
func CheckCorpus(code string) CorpusReport {
	r := CorpusReport{Code: code}

	m, ok := Meta(code)
	if !ok || m.Lang == "" || m.Name == "" || m.PsalmSystem == "" {
		r.Errors = append(r.Errors, "missing or incomplete sidecar (need lang, name, psalm_system)")
	}
	if m.PsalmSystem != "" && !validPsalmSystems[m.PsalmSystem] {
		r.Errors = append(r.Errors, fmt.Sprintf("invalid psalm_system %q (want vulgate|hebrew|drb)", m.PsalmSystem))
	}

	c := load(code)
	if len(c.books) == 0 {
		r.Errors = append(r.Errors, "no verses parsed (empty or malformed .tsv)")
		sort.Strings(r.Errors)
		return r
	}

	canon := CanonicalBooks()
	for book := range c.books {
		if !canon[book] {
			r.Errors = append(r.Errors, "unknown book name: "+book)
		}
	}

	ref := load("vul")
	for book, chaps := range c.books {
		for ch, verses := range chaps {
			seen := map[int]bool{}
			maxV := 0
			for _, v := range verses {
				if seen[v.Verse] {
					r.Warnings = append(r.Warnings, fmt.Sprintf("%s %d: duplicate verse %d", book, ch, v.Verse))
				}
				seen[v.Verse] = true
				if v.Verse > maxV {
					maxV = v.Verse
				}
			}
			if maxV > len(seen) {
				r.Warnings = append(r.Warnings, fmt.Sprintf("%s %d: verse gap (have %d verse(s), highest numbered %d)", book, ch, len(seen), maxV))
			}
		}
		if refChaps, ok := ref.books[book]; ok {
			for ch := range refChaps {
				if _, have := chaps[ch]; !have {
					r.Warnings = append(r.Warnings, fmt.Sprintf("%s: missing chapter %d (present in vul)", book, ch))
				}
			}
		}
	}

	sort.Strings(r.Errors)
	sort.Strings(r.Warnings)
	return r
}