1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
|
module B = Colitur_citation.Book
let s = Alcotest.string
let test_both_spellings_are_one_book () =
(* The books listed here arrive in more than one spelling and must
collapse onto one id. This is the whole reason the parser exists rather
than a regex. Most pairs are dotted/undotted; "Sir"/"Ecclus" and
"Apoc"/"Rev" are a modern spelling sitting inside otherwise-Vulgate
data -- see book.mli. *)
let same a b =
match B.of_token a, B.of_token b with
| Some x, Some y ->
Alcotest.(check s) (a ^ " = " ^ b) (B.to_string x) (B.to_string y)
| _ -> Alcotest.failf "%s or %s did not resolve" a b
in
same "Isa" "Isa.";
same "Matt" "Matt.";
same "1 Cor" "1 Cor.";
same "1 Pet" "1 Pet.";
same "1 Thess" "1 Thess.";
same "Eph" "Eph.";
same "3 Kgs." "3 Kings";
same "Sir" "Ecclus";
same "Apoc" "Rev";
same "2 Cor" "2 Cor.";
same "Col" "Col.";
same "Wis" "Wis.";
same "2 Tim" "2 Tim.";
same "Ex" "Exod";
same "Ezech" "Ezek";
same "Jas" "James"
(* The ordinal spellings resolve in the token table. NOTE: this does NOT
exercise [Parse.split_book]'s leading-digit scan -- of_token is a plain
table lookup. The parse-layer cases in [parse_suite] cover that; both are
needed, and confusing the two is how the gap survived review once already. *)
let test_ordinal_books_beyond_one () =
let id t =
match B.of_token t with
| Some x -> B.to_string x
| None -> Alcotest.failf "%s did not resolve" t
in
Alcotest.(check s) "3 Kings" "kings_3" (id "3 Kings");
Alcotest.(check s) "3 Kgs." "kings_3" (id "3 Kgs.");
Alcotest.(check s) "4 Kings" "kings_4" (id "4 Kings")
(* The fallback display name: what a book renders as when a language file has
no [bible] entry for it. Must be the data's own spelling, never the id --
and must itself re-parse, or Render/Parse round-tripping breaks. *)
let test_default_spelling () =
let sp t =
match B.of_token t with
| Some x -> B.default_spelling x
| None -> Alcotest.failf "%s did not resolve" t
in
Alcotest.(check s) "luke" "Luke" (sp "Luke");
Alcotest.(check s) "kings_3 uses first spelling" "3 Kings" (sp "3 Kgs.");
let cited = List.map snd B.tokens in
List.iter
(fun id ->
if List.mem id cited then begin
let d = B.default_spelling id in
match B.of_token d with
| Some back when B.to_string back = B.to_string id -> ()
| _ ->
Alcotest.failf "default_spelling %s = %S does not resolve back"
(B.to_string id) d
end)
B.all
let test_unknown_token_is_none () =
Alcotest.(check bool) "not a book" true (B.of_token "Nonesuch" = None)
let test_vulgate_is_identity () =
match B.of_token "3 Kings" with
| None -> Alcotest.fail "3 Kings did not resolve"
| Some k ->
Alcotest.(check s) "unmapped" "kings_3" (B.to_string (B.map B.vulgate k))
let test_modern_renumbers () =
let modern = B.tradition_of_fields [ ("kings_3", "kings_1") ] in
match B.of_token "3 Kings" with
| None -> Alcotest.fail "3 Kings did not resolve"
| Some k ->
Alcotest.(check s) "renumbered" "kings_1" (B.to_string (B.map modern k))
let test_tokens_has_no_duplicate_spelling () =
(* [of_token] resolves via [List.assoc_opt], which silently prefers the
first match on a duplicate key. A copy-paste collision in the table
would therefore mis-map a book in total silence, never an exception --
assert the invariant directly rather than trust it by inspection. *)
let spellings = List.map fst B.tokens in
let sorted = List.sort compare spellings in
let rec find_dup = function
| a :: (b :: _ as rest) -> if a = b then Some a else find_dup rest
| _ -> None
in
match find_dup sorted with
| None -> ()
| Some dup -> Alcotest.failf "duplicate spelling in Book.tokens: %S" dup
(* Read a whole file into a string. Test-only I/O; the library itself never
touches the filesystem. *)
let read_file path =
let ic = open_in_bin path in
let n = in_channel_length ic in
let content = really_input_string ic n in
close_in ic;
content
let starts_with_at content pos prefix =
let plen = String.length prefix in
pos + plen <= String.length content && String.sub content pos plen = prefix
(* Every [(reference ...)] payload in a data file, in file order. No
[Str]/regex -- a plain forward scan for the marker, then read to the
closing quote. Mirrors the coordinator's own survey command (grep -oh
over the reference marker, quote-delimited). *)
let references_with marker content =
let mlen = String.length marker in
let len = String.length content in
let rec loop pos acc =
if pos >= len then List.rev acc
else if starts_with_at content pos marker then
let start = pos + mlen in
match String.index_from_opt content start '"' with
| None -> List.rev acc
| Some close ->
let payload = String.sub content start (close - start) in
loop (close + 1) (payload :: acc)
else loop (pos + 1) acc
in
loop 0 []
let is_alpha c = (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z')
let is_one_to_four c = c >= '1' && c <= '4'
(* The leading book token of one reference payload, e.g. ["2 Tim 4:1-8"] ->
["2 Tim"], ["Wis. 5:1-5"] -> ["Wis."]. Mirrors the coordinator's own
survey command's second stage:
`sed -E 's/^(([1-4] )?[A-Za-z]+\.?).*/\1/'`. *)
let book_token r =
let len = String.length r in
let start = if len >= 2 && is_one_to_four r.[0] && r.[1] = ' ' then 2 else 0 in
let i = ref start in
while !i < len && is_alpha r.[!i] do
incr i
done;
let stop = if !i < len && r.[!i] = '.' then !i + 1 else !i in
String.sub r 0 stop
(* A citation is written TWO ways in this project's data, and a survey that
knows only one of them silently under-reads the corpus:
(reference "Isa 60:1-6") -- lectionary/sanctoral/commons
(Set_citation First "Wis 7:7-14") -- adjustments.sexp, an overlay
Both markers are scanned below. *)
let references content =
(* Each marker must include the OPENING QUOTE. [references_with] returns
the text between the marker and the next '"', so a marker stopping at
"(Set_citation " yields "First " / "Gospel " -- the part label, never
the citation -- and every entry is then silently discarded. That was
this function's first version, and it read the file while extracting
nothing at all. *)
references_with "(reference \"" content
@ references_with "(Set_citation First \"" content
@ references_with "(Set_citation Gospel \"" content
let test_every_data_file_token_resolves () =
(* The check whose absence caused fix round 1: the brief surveyed only
the lectionary and missed 21 tokens, several common, living in
sanctoral.sexp and commons.sexp.
adjustments.sexp was then missed AGAIN, by this very test, because it
writes citations as `Set_citation` rather than `(reference ...)` -- the
same "one file too few" shape twice over. All FOUR shipped files are
read here, and the token set is re-derived from them at test time
rather than hardcoded, so this keeps working when the data changes. *)
let files =
[ "../data/ef/lectionary.sexp"; "../data/ef/sanctoral.sexp";
"../data/ef/commons.sexp"; "../data/ef/adjustments.sexp" ]
in
let tokens =
files
|> List.concat_map (fun f -> references (read_file f))
|> List.map book_token
|> List.sort_uniq compare
in
Alcotest.(check bool) "at least one token found" true (List.length tokens > 0);
let unresolved = List.filter (fun t -> B.of_token t = None) tokens in
match unresolved with
| [] -> ()
| _ ->
Alcotest.failf "unresolved book tokens in shipped data: %s"
(String.concat ", " unresolved)
(* lang/traditions.ini's own [modern] section, duplicated here on purpose:
the shipped file and this expectation are written twice and must agree,
so a typo in either is visible. Two ids mapping onto the SAME target is
always a typo -- it would silently merge two books -- so this asserts the
mapping is injective, not merely that every id is known. *)
let test_modern_tradition_is_injective () =
(* READ the shipped file. An earlier version of this test hand-copied the
table, which asserted things about the copy and nothing at all about the
artifact -- a mapping added to traditions.ini and forgotten here would
have passed. *)
let text =
let ic = open_in_bin "../lang/traditions.ini" in
let s = really_input_string ic (in_channel_length ic) in
close_in ic; s
in
let sections =
match Colitur_kernel.Overlay_ini.parse_sections text with
| Ok ss -> ss
| Error e -> Alcotest.failf "lang/traditions.ini: %s" e
in
let section name =
List.filter_map
(fun (sc : Colitur_kernel.Overlay_ini.section) ->
if sc.Colitur_kernel.Overlay_ini.name = name then
Some sc.Colitur_kernel.Overlay_ini.fields
else None)
sections
|> List.concat
in
(* [vulgate] must be the identity: an entry there would silently renumber
the DEFAULT, which is the one thing this design promises never happens. *)
Alcotest.(check (list (pair string string))) "vulgate is identity" []
(section "vulgate");
let fields = section "modern" in
Alcotest.(check bool) "modern is not empty" true (fields <> []);
Alcotest.(check (list string)) "all ids known" [] (B.unknown_fields fields);
(* every declared tradition target is actually reachable -- an unused target
is either a forgotten mapping or dead weight *)
let targets = List.map snd fields in
(* [B.tokens] is (spelling, id) pairs, so membership of an ID is a lookup
over the VALUES, not [mem_assoc] over the keys. Getting that backwards
flags every ordinary book as an unused target. *)
let cited = List.map (fun (_, i) -> B.to_string i) B.tokens in
List.iter
(fun id ->
let n = B.to_string id in
if (not (List.mem n cited)) && not (List.mem n targets) then
Alcotest.failf "%s is a declared target no tradition maps onto" n)
B.all;
let uniq = List.sort_uniq compare targets in
Alcotest.(check int) "no two ids share a target"
(List.length targets) (List.length uniq)
let suite =
( "book",
[ Alcotest.test_case "both spellings one book" `Quick test_both_spellings_are_one_book;
Alcotest.test_case "ordinal books 3 and 4" `Quick test_ordinal_books_beyond_one;
Alcotest.test_case "default spelling round-trips" `Quick test_default_spelling;
Alcotest.test_case "unknown token" `Quick test_unknown_token_is_none;
Alcotest.test_case "vulgate identity" `Quick test_vulgate_is_identity;
Alcotest.test_case "modern renumbers" `Quick test_modern_renumbers;
Alcotest.test_case "tokens has no duplicate spelling" `Quick
test_tokens_has_no_duplicate_spelling;
Alcotest.test_case "every data file token resolves" `Quick
test_every_data_file_token_resolves;
Alcotest.test_case "modern tradition is injective" `Quick
test_modern_tradition_is_injective ] )
module P = Colitur_citation.Parse
(* Render a parse back to a debug string so a test can assert shape
compactly: "book|chapter:v-v,v-v|chapter:v". *)
let show (t : P.t) =
let range (r : P.verse_range) =
match r.P.last with
| None -> string_of_int r.P.first
| Some l -> Printf.sprintf "%d-%d" r.P.first l
in
let part (p : P.part) =
Printf.sprintf "%d:%s" p.P.chapter
(String.concat "," (List.map range p.P.verses))
in
B.to_string t.P.book ^ "|" ^ String.concat "|" (List.map part t.P.parts)
let parses input expected () =
match P.parse input with
| Error e -> Alcotest.failf "%s did not parse: %s" input e
| Ok t -> Alcotest.(check s) input expected (show t)
let test_rejects_unknown_book () =
Alcotest.(check bool) "error" true (Result.is_error (P.parse "Nonesuch 1:1"))
let test_rejects_garbage () =
List.iter
(fun bad ->
Alcotest.(check bool) bad true (Result.is_error (P.parse bad)))
[ ""; "Luke"; "Luke :"; "Luke 1:"; "Luke abc:1" ]
let parse_suite =
[ ("simple", `Quick, parses "1 Cor 11:20-32" "corinthians_1|11:20-32");
("trailing period", `Quick, parses "1 John 3:13-18." "john_1|3:13-18");
("single verse", `Quick, parses "Luke 2:21" "luke|2:21");
("dotted spelling", `Quick, parses "Isa. 1:16-19" "isaiah|1:16-19");
("verse list", `Quick, parses "Acts 10:34, 42-48" "acts|10:34,42-48");
("new chapter", `Quick, parses "1 Cor. 9:24-27; 10:1-5"
"corinthians_1|9:24-27|10:1-5");
(* Rule 1: the second part names no chapter, so it inherits chapter 2. *)
("inherited chapter", `Quick, parses "Joel 2:23-24; 26-27" "joel|2:23-24|2:26-27");
(* Rule 2: comma separates chapter from verses in the second part. *)
("comma chapter", `Quick, parses "Mark 14:32-72; 15, 1-46"
"mark|14:32-72|15:1-46");
("four ranges", `Quick, parses "Dan 13:1-9, 15-17, 19-30, 33-62."
"daniel|13:1-9,15-17,19-30,33-62");
("mixed", `Quick, parses "Num 20:1, 3; 6-13." "numbers|20:1,3|20:6-13");
("trailing semicolon", `Quick, parses "1 Cor 1:18-25; 1:30;"
"corinthians_1|1:18-25|1:30");
("four parts", `Quick, parses "Eccli 24:5; 14:7; 14:9-11; 24:30-31"
"ecclesiasticus|24:5|14:7|14:9-11|24:30-31");
("modern name, vulgate id", `Quick, parses "Rev 12:1" "apocalypse|12:1");
(* Ordinals 3 and 4 must survive [split_book]'s leading-digit scan.
Narrowing its '1'..'4' range to '1'..'3' passes every OTHER case in
this suite silently, while seven real citations depend on it. A
Book.of_token test does NOT cover this -- that is a table lookup and
never reaches split_book. *)
("ordinal 3 parses", `Quick, parses "3 Kings 17:8-16" "kings_3|17:8-16");
("ordinal 3 dotted", `Quick, parses "3 Kgs. 19:3-8" "kings_3|19:3-8");
("ordinal 4 parses", `Quick, parses "4 Kings 5:1-15" "kings_4|5:1-15");
("unknown book", `Quick, test_rejects_unknown_book);
("garbage", `Quick, test_rejects_garbage) ]
module R = Colitur_citation.Render
let names id form =
let n = B.to_string id in
match form with `Abbr -> (if n = "luke" then "Luc." else n)
| `Full -> (if n = "luke" then "Evangelium secundum Lucam" else n)
let render_with fields input expected () =
let style = R.style_of_fields fields in
match P.parse input with
| Error e -> Alcotest.failf "%s did not parse: %s" input e
| Ok t -> Alcotest.(check s) input expected (R.render style ~names t)
let test_unquotes_trailing_space () =
let st = R.style_of_fields [ ("part_sep", "\"; \"") ] in
Alcotest.(check s) "quotes stripped, space kept" "; " (R.part_sep st)
let test_bare_value_untouched () =
let st = R.style_of_fields [ ("range", "{first}-{last}") ] in
Alcotest.(check s) "no quotes" "{first}-{last}" (R.range st)
(* [subst] has no test pressure of its own anywhere else in the suite, so
these three isolate its contract directly through the only surface that
calls it: a [range]/[chapter_verse] template chosen so nothing else in
the pipeline (verse-list joining, book naming) can mask the result. *)
let test_subst_no_placeholder () =
(* A template with no "{" at all passes through completely unchanged. *)
render_with [ ("book", "abbr"); ("chapter_verse", "fixed-text") ] "Luke 2:21"
"Luc. fixed-text" ()
let test_subst_two_placeholders () =
(* Two DIFFERENT known placeholders, reordered relative to the record's
own field order ([last] before [first]) -- proves substitution is by
name, not by position. *)
render_with
[ ("book", "abbr"); ("range", "{last}~{first}") ]
"Luke 5:12-14" "Luc. 5:14~12" ()
let test_subst_unknown_placeholder_alone () =
(* A template that is NOTHING but an unrecognised placeholder: it must
survive byte-for-byte, proving [subst] never touches an unmatched
"{...}" run even when there is no surrounding literal text to anchor
on. *)
render_with [ ("book", "abbr"); ("range", "{nope}") ] "Luke 5:12-14"
"Luc. 5:{nope}" ()
(* [roman_numeral] is private to Render (not in the .mli, matching
[View.roman_numeral]'s own precedent -- tested indirectly, never
exposed). Reached through [render] with a [chapter_verse] template of
bare [{chapter_roman}], [book_sep] emptied and [names] returning "", so
the assertion isolates exactly the numeral and nothing else. *)
let roman_of n =
let luke = match B.of_token "Luke" with Some id -> id | None -> assert false in
let t : P.t = { P.book = luke; parts = [ { P.chapter = n; verses = [ { P.first = 1; last = None } ] } ] } in
let style = R.style_of_fields [ ("book_sep", ""); ("chapter_verse", "{chapter_roman}") ] in
R.render style ~names:(fun _ _ -> "") t
let test_roman_numeral_values () =
List.iter
(fun (n, expected) -> Alcotest.(check s) (string_of_int n) expected (roman_of n))
[ (1, "I"); (4, "IV"); (9, "IX"); (14, "XIV"); (40, "XL"); (150, "CL") ]
let test_roman_numeral_non_positive () =
(* The one input that could loop forever in a naive implementation: a
non-positive chapter returns the arabic form instead. Completing at
all is part of what this test proves. *)
Alcotest.(check s) "zero" "0" (roman_of 0);
Alcotest.(check s) "negative" "-3" (roman_of (-3))
let render_suite =
( "Citation/render",
[ Alcotest.test_case "latin default" `Quick
(render_with [ ("book", "abbr") ] "Luke 5:12-14" "Luc. 5:12-14");
Alcotest.test_case "full name" `Quick
(render_with [ ("book", "full") ] "Luke 5:12-14"
"Evangelium secundum Lucam 5:12-14");
Alcotest.test_case "comma style" `Quick
(render_with
[ ("book", "abbr"); ("chapter_verse", "{chapter}, {verses}") ]
"Luke 5:12-14" "Luc. 5, 12-14");
Alcotest.test_case "multi part" `Quick
(render_with [ ("book", "abbr"); ("part_sep", "\"; \"") ]
"Joel 2:23-24; 26-27" "joel 2:23-24; 2:26-27");
Alcotest.test_case "unquote" `Quick test_unquotes_trailing_space;
Alcotest.test_case "bare value" `Quick test_bare_value_untouched;
(* The Missal's own convention, scan-verified: "Matth. 11, 2" /
"Ioann. 1, 1" -- dotted abbreviation, comma, ARABIC chapter. *)
Alcotest.test_case "missal convention" `Quick
(render_with
[ ("book", "abbr"); ("chapter_verse", "{chapter}, {verses}") ]
"Luke 5:12-14" "Luc. 5, 12-14");
Alcotest.test_case "roman chapter" `Quick
(render_with
[ ("book", "abbr"); ("chapter_verse", "{chapter_roman}, {verses}") ]
"Luke 5:12-14" "Luc. V, 12-14");
(* U+00A0 between book and reference, for typeset output. *)
Alcotest.test_case "nbsp book sep" `Quick
(render_with [ ("book", "abbr"); ("book_sep", "\"\xc2\xa0\"") ]
"Luke 5:12-14" "Luc.\xc2\xa05:12-14");
Alcotest.test_case "unknown placeholder survives" `Quick
(render_with
[ ("book", "abbr"); ("chapter_verse", "{chapter}:{nope}") ]
"Luke 5:12-14" "Luc. 5:{nope}");
Alcotest.test_case "subst: no placeholder" `Quick test_subst_no_placeholder;
Alcotest.test_case "subst: two placeholders" `Quick test_subst_two_placeholders;
Alcotest.test_case "subst: unknown placeholder alone" `Quick
test_subst_unknown_placeholder_alone;
Alcotest.test_case "roman_numeral: table values" `Quick test_roman_numeral_values;
Alcotest.test_case "roman_numeral: non-positive" `Quick
test_roman_numeral_non_positive ] )
module S = Colitur_citation.Sigla
let test_verbatim_is_identity () =
List.iter
(fun c -> Alcotest.(check s) c c (S.format S.verbatim c))
[ "3 Kgs. 19:3-8"; "Isa. 1:16-19"; "not a citation at all" ]
let test_unparseable_survives_unchanged () =
let sg = S.make ~style:R.default_style ~tradition:B.vulgate ~names in
Alcotest.(check s) "passed through" "Nonesuch 1:1" (S.format sg "Nonesuch 1:1")
let test_tradition_applies () =
let modern = B.tradition_of_fields [ ("kings_3", "kings_1") ] in
let sg = S.make ~style:R.default_style ~tradition:modern ~names in
Alcotest.(check s) "renumbered" "kings_1 19:3-8" (S.format sg "3 Kings 19:3-8")
let sigla_suite =
[ ("verbatim identity", `Quick, test_verbatim_is_identity);
("unparseable survives", `Quick, test_unparseable_survives_unchanged);
("tradition applies", `Quick, test_tradition_applies) ]
|