aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--lib/citation/book.ml107
-rw-r--r--lib/citation/book.mli20
-rw-r--r--lib/citation/parse.ml105
-rw-r--r--lib/citation/parse.mli9
-rw-r--r--lib/citation/render.ml8
-rw-r--r--test/test_citation.ml9
-rw-r--r--test/test_citation_coverage_of.ml140
-rw-r--r--test/test_colitur.ml1
8 files changed, 350 insertions, 49 deletions
diff --git a/lib/citation/book.ml b/lib/citation/book.ml
index 986a898..b43cdd8 100644
--- a/lib/citation/book.ml
+++ b/lib/citation/book.ml
@@ -19,26 +19,26 @@ let table =
("kings_3", [ "3 Kings"; "3 Kgs."; "3 Reg"; "Liber Regum III" ]);
("kings_4", [ "4 Kings"; "4 Reg"; "Liber Regum IV"; "4 Kgs." ]);
("esdras_2", [ "2 Esd."; "2 Esdr"; "Liber Esdrae"; "2 Esdras" ]);
- ("tobit", [ "Tob"; "Liber Tobiae"; "Tobias" ]);
+ ("tobit", [ "Tob"; "Liber Tobiae"; "Tobias"; "Tobit" ]);
("judith", [ "Judith"; "Iudith"; "Liber Iudith"; "Jth" ]);
("esther", [ "Esther"; "Esth"; "Liber Esther" ]);
("proverbs", [ "Prov"; "Proverbs" ]);
- ("song_of_songs", [ "Song"; "Cant."; "Canticle of Canticles" ]);
+ ("song_of_songs", [ "Song"; "Cant."; "Canticle of Canticles"; "Song of Solomon" ]);
("wisdom", [ "Wis"; "Wis."; "Sap"; "Liber Sapientiae"; "Wisdom" ]);
("ecclesiasticus", [ "Ecclus"; "Sir"; "Eccli"; "Ecclesiasticus" ]);
- ("isaiah", [ "Isa"; "Isa."; "Isai"; "Isaias Propheta"; "Isaias" ]);
- ("jeremiah", [ "Jer"; "Ier"; "Ieremias Propheta"; "Jeremias" ]);
- ("ezekiel", [ "Ezech"; "Ezek"; "Ezechiel Propheta"; "Ezechiel" ]);
+ ("isaiah", [ "Isa"; "Isa."; "Isai"; "Isaias Propheta"; "Isaias"; "Isaiah" ]);
+ ("jeremiah", [ "Jer"; "Ier"; "Ieremias Propheta"; "Jeremias"; "Jeremiah" ]);
+ ("ezekiel", [ "Ezech"; "Ezek"; "Ezechiel Propheta"; "Ezechiel"; "Ezekiel" ]);
("daniel", [ "Dan"; "Daniel Propheta"; "Daniel" ]);
("osee", [ "Osee"; "Osee Propheta" ]);
("joel", [ "Joel"; "Ioel"; "Ioel Propheta" ]);
("jonas", [ "Jonas"; "Ionae"; "Ionas Propheta" ]);
- ("malachi", [ "Mal"; "Malach"; "Malachias Propheta"; "Malachias" ]);
- ("matthew", [ "Matt"; "Matt."; "Matth"; "Evangelium secundum Matthaeum"; "Matthew" ]);
+ ("malachi", [ "Mal"; "Malach"; "Malachias Propheta"; "Malachias"; "Malachi" ]);
+ ("matthew", [ "Matt"; "Matt."; "Matth"; "Mat"; "Evangelium secundum Matthaeum"; "Matthew" ]);
("mark", [ "Mark"; "Marc"; "Evangelium secundum Marcum" ]);
("luke", [ "Luke"; "Luc"; "Evangelium secundum Lucam" ]);
("john", [ "John"; "Ioann"; "Evangelium secundum Ioannem" ]);
- ("acts", [ "Acts"; "Act"; "Actus Apostolorum"; "Acts of the Apostles" ]);
+ ("acts", [ "Acts"; "Act"; "Actus Apostolorum"; "Acts of the Apostles"; "The Acts" ]);
("romans", [ "Rom"; "Epistola ad Romanos"; "Romans" ]);
("corinthians_1", [ "1 Cor"; "1 Cor."; "Epistola I ad Corinthios"; "1 Corinthians" ]);
("corinthians_2", [ "2 Cor"; "2 Cor."; "Epistola II ad Corinthios"; "2 Corinthians" ]);
@@ -64,6 +64,73 @@ let table =
an [of_token] entry would let one book carry two different ids. *)
("apocalypse", [ "Apoc"; "Rev"; "Liber Apocalypsis"; "Apocalypse" ]) ]
+(* Fix wave I2 (final-review.md, 2026-08-25-colitur-of-phases-3-5): books
+ data/of/lectionary.sexp cites that no EF file ever needed, so the survey
+ above never registered them -- 345 of 730 OF citation fields printed
+ unconverted (35%) because [Book.of_token] had never heard of them, not
+ because [Parse]'s grammar itself could not read the reference. English-
+ canonical spellings only, verified directly against that one file
+ (`grep`, not invented) -- the SAME "every spelling any of the files uses"
+ discipline [table] above states, extended to a fourth file.
+
+ Deliberately a SEPARATE table, not folded into [table] above, for one
+ reason that matters: [table]'s ids are covered by [all] below, and
+ book.mli's own contract for [all] is "every SHIPPED language file is
+ asserted complete over" it (test_lang_coverage.ml's [test_every_book_
+ named] and its siblings). lang/la.ini's own header is equally strict --
+ "do not invent a Latin title", every `.full`/`.abbr` row cited to an
+ exact scan line -- and NONE of these 25 books is attested anywhere in
+ the EF's own scans (the EF lectionary/propers/commons never read from
+ Ruth, Job, Judges, Philemon, ... at all). Registering them in [table]
+ would force a choice between fabricating 50 uncited Latin titles (the
+ one thing this project refuses hardest) and quietly weakening book.mli's
+ own stated promise for the ids that DO have one. Neither is right, so
+ these ids are resolvable ([of_token]/[tokens]/[default_spelling] below
+ all include them) but excluded from [all] -- a citation using one still
+ PARSES and RENDERS (bin/main.ml's own [names_of] already falls back to
+ {!default_spelling} on a lang-table miss, book.mli's own documented
+ reason that function exists), it just renders in its own English
+ spelling rather than a Latin one no primary source has ever supplied,
+ the exact "degraded, not broken" contract {!default_spelling} documents.
+
+ [samuel_1]/[samuel_2] in particular are NOT [kings_1]/[kings_2] below:
+ those tradition targets already denote (via traditions.ini's own
+ [modern] mapping, "kings_3 = kings_1") the SAME physical book as the
+ Vulgate's own [kings_3]/[kings_4] ("3/4 Kings" = modern "1/2 Kings") --
+ a different pair of books from Samuel ("Liber Regum I/II" in the
+ Vulgate's own naming, never cited by any shipped EF data and so never
+ given an id before now). *)
+let of_lectionary_table =
+ [ ("samuel_1", [ "1 Sam"; "1 Samuel" ]);
+ ("samuel_2", [ "2 Sam"; "2 Samuel" ]);
+ ("joshua", [ "Josh"; "Joshua" ]);
+ ("judges", [ "Judg"; "Judges" ]);
+ ("ruth", [ "Ruth" ]);
+ ("deuteronomy", [ "Deut"; "Deuteronomy" ]);
+ ("chronicles_1", [ "1 Chron"; "1 Chronicles" ]);
+ ("chronicles_2", [ "2 Chron"; "2 Chronicles" ]);
+ ("ezra", [ "Ezra" ]);
+ ("job", [ "Job" ]);
+ ("ecclesiastes", [ "Eccl"; "Ecclesiastes" ]);
+ ("baruch", [ "Bar"; "Baruch" ]);
+ ("lamentations", [ "Lam"; "Lamentations" ]);
+ ("amos", [ "Amos" ]);
+ ("micah", [ "Mic"; "Micah" ]);
+ ("nahum", [ "Nah"; "Nahum" ]);
+ ("habakkuk", [ "Hab"; "Habakkuk" ]);
+ ("zephaniah", [ "Zeph"; "Zephaniah" ]);
+ ("haggai", [ "Hag"; "Haggai" ]);
+ ("zechariah", [ "Zech"; "Zechariah" ]);
+ ("maccabees_1", [ "1 Macc"; "1 Maccabees" ]);
+ ("maccabees_2", [ "2 Macc"; "2 Maccabees" ]);
+ ("philemon", [ "Phlm"; "Philemon" ]);
+ ("jude", [ "Jude" ]);
+ (* [john_2]/[john_3], siblings of [john_1] above -- three separate
+ Johannine epistles, three separate ids, the same one-id-per-physical-
+ book rule every other entry in this file follows. *)
+ ("john_2", [ "2 John" ]);
+ ("john_3", [ "3 John" ]) ]
+
(* Targets a tradition can map ONTO that the Vulgate data never cites
directly. Present so [tradition_of_fields] can validate both sides.
[sirach] and [revelation] exist ONLY here, never as an [of_token] result:
@@ -89,15 +156,20 @@ let tradition_target_table =
let tradition_targets = List.map fst tradition_target_table
+(* Deliberately NOT [of_lectionary_table] -- see that table's own header for
+ why: those ids have no la.ini-verified name, so they stay out of the set
+ {!Colitur_naming.Lang}-completeness checks (test_lang_coverage.ml) run
+ against, without weakening what those checks still guarantee for every id
+ here. *)
let all = List.map fst table @ tradition_targets
let tokens =
List.concat_map
(fun (id, sp) -> List.map (fun s -> (s, id)) sp)
- (table @ tradition_target_table)
+ (table @ tradition_target_table @ of_lectionary_table)
let default_spelling id =
- match List.assoc_opt id (table @ tradition_target_table) with
+ match List.assoc_opt id (table @ tradition_target_table @ of_lectionary_table) with
| Some (first :: _) -> first
| Some [] | None -> id
@@ -120,3 +192,18 @@ let unknown_fields fields =
fields
let map tr id = match List.assoc_opt id tr with Some x -> x | None -> id
+
+(* Fix wave I2: books with exactly one chapter are conventionally cited by
+ VERSE ALONE, no chapter number at all ("Jude 17, 20b-25", "Phlm 7-22") --
+ found while adding [of_lectionary_table] above: [jude]'s own citation
+ "Jude 17,20b-25" used to fail as "unknown book" (safe, if unconverted);
+ registering [jude] made it PARSE, but WRONGLY -- [Parse]'s existing
+ "a leading comma-piece that looks like a number introduces a chapter"
+ rule (parse.ml's own [parse_part], written for real multi-chapter data
+ like "15, 1-46") misread "17" as chapter 17, which does not exist in a
+ 25-verse, one-chapter book. A parse that SUCCEEDS with the wrong
+ structure is worse than one that fails: nothing downstream would ever
+ suspect it. [Parse.parse] consults this to skip that rule entirely for
+ these four ids and default to chapter 1. *)
+let single_chapter_ids = [ "jude"; "philemon"; "john_2"; "john_3" ]
+let is_single_chapter id = List.mem id single_chapter_ids
diff --git a/lib/citation/book.mli b/lib/citation/book.mli
index 0ca229c..4e91c6c 100644
--- a/lib/citation/book.mli
+++ b/lib/citation/book.mli
@@ -30,7 +30,18 @@ val to_string : id -> string
tradition-only targets [sirach]/[revelation] -- see those below. *)
val of_token : string -> id option
-(** Every id this build knows, for coverage checks. *)
+(** Every id surveyed against the EF's own three citation-bearing files
+ (lectionary, sanctoral propers, commons) -- for coverage checks: every
+ SHIPPED language file's [\[bible\]] section is asserted complete over
+ this list (test_lang_coverage.ml). NOT every id {!of_token} can resolve:
+ book.ml's own [of_lectionary_table] adds a further ~25 ids the OF's
+ English-canonical lectionary cites that no EF file ever needed and that
+ no primary EF source attests a Latin title for -- resolvable ([of_token]/
+ {!default_spelling} both cover them) but deliberately excluded here,
+ rather than either fabricating an uncited Latin name or weakening this
+ list's own completeness promise for the ids that DO have one. Such a
+ book still renders, in its own English spelling ({!default_spelling}'s
+ own fallback), never a raw internal id. *)
val all : id list
(** Every accepted spelling paired with its id. *)
@@ -77,3 +88,10 @@ val tradition_of_fields : (string * string) list -> tradition
val unknown_fields : (string * string) list -> string list
val map : tradition -> id -> id
+
+(** Whether [id] names a book with exactly one chapter (Jude, Philemon,
+ 2 John, 3 John), conventionally cited by verse alone with no chapter
+ number ("Jude 17", not "Jude 1:17"). {!Colitur_citation.Parse.parse}
+ consults this to avoid misreading a leading verse number as a chapter
+ -- see book.ml's own citation for the real citation this was found on. *)
+val is_single_chapter : id -> bool
diff --git a/lib/citation/parse.ml b/lib/citation/parse.ml
index 2f2316e..2c73600 100644
--- a/lib/citation/parse.ml
+++ b/lib/citation/parse.ml
@@ -1,6 +1,19 @@
(* SPDX-License-Identifier: AGPL-3.0-or-later *)
-type verse_range = { first : int; last : int option }
+(* Fix wave I2 (final-review.md, 2026-08-25-colitur-of-phases-3-5): a VERSE
+ number (never a chapter number -- see [num_opt]'s own citation below) may
+ carry a trailing lowercase-letter sub-verse marker, standard lectionary
+ notation for "part of this verse" ("11a" = the first clause of verse 11)
+ and sometimes several run together ("1bcde" = parts b through e of verse
+ 1). [suffix] is [""] for the overwhelming majority of verse numbers --
+ every EF citation, checked: none carries one at all (grep across
+ data/ef/{lectionary,sanctoral,commons}.sexp finds zero) -- so this is
+ purely additive for EF and does not change how any existing citation
+ parses or renders. Carried through and RENDERED, not stripped: dropping
+ it would silently lose real precision a reader can see in the source
+ text, trading "wrong format, complete" for "right format, incomplete". *)
+type verse_num = { n : int; suffix : string }
+type verse_range = { first : verse_num; last : verse_num option }
type part = { chapter : int; verses : verse_range list }
type t = { book : Book.id; parts : part list }
@@ -75,15 +88,38 @@ let int_opt s =
if not ok then None
else match int_of_string_opt s with Some n when n > 0 -> Some n | _ -> None
+(* [num_opt] is [int_opt] widened to accept a trailing sub-verse letter run
+ ([verse_num]'s own citation above) -- used ONLY for verse numbers inside
+ [parse_range] below. Chapter numbers ([parse_part]'s own [int_opt c]
+ calls) are deliberately left on plain [int_opt]: nothing in the data ever
+ attaches a sub-verse letter to a CHAPTER, and keeping chapter-parsing on
+ the original, narrower function means this change cannot loosen what a
+ chapter number is allowed to look like. *)
+let split_num_suffix s =
+ let n = String.length s in
+ let i = ref 0 in
+ while !i < n && s.[!i] >= '0' && s.[!i] <= '9' do incr i done;
+ if !i = 0 then None
+ else
+ let digits = String.sub s 0 !i and suffix = String.sub s !i (n - !i) in
+ if String.for_all (fun c -> c >= 'a' && c <= 'z') suffix then Some (digits, suffix) else None
+
+let num_opt s : verse_num option =
+ match split_num_suffix (String.trim s) with
+ | None -> None
+ | Some (digits, suffix) -> ( match int_opt digits with Some n -> Some { n; suffix } | None -> None)
+
(* "20-32" -> {first=20; last=Some 32}; "21" -> {first=21; last=None} *)
let parse_range s =
match split_on '-' s with
- | [ a ] -> ( match int_opt a with Some f -> Some { first = f; last = None } | None -> None)
+ | [ a ] -> ( match num_opt a with Some f -> Some { first = f; last = None } | None -> None)
| [ a; b ] -> (
- match (int_opt a, int_opt b) with
+ match (num_opt a, num_opt b) with
(* A descending range ("1:20-10") is always a transcription error;
- accepting it would render back out as a citation nobody can follow. *)
- | Some f, Some l when l >= f -> Some { first = f; last = Some l }
+ accepting it would render back out as a citation nobody can follow.
+ Compared on the NUMBER only -- "11a-11b" is a real, ascending
+ sub-verse range even though nothing here orders letters. *)
+ | Some f, Some l when l.n >= f.n -> Some { first = f; last = Some l }
| _ -> None)
| _ -> None
@@ -97,35 +133,45 @@ let parse_ranges s =
pieces (Some [])
(* One ";"-separated part. [inherited] is the chapter of the previous part,
- used when this one names none (rule 1 in the grammar table). *)
-let parse_part ~inherited s =
+ used when this one names none (rule 1 in the grammar table).
+ [single_chapter] ({!Book.is_single_chapter}, fix wave I2) skips rule 3
+ entirely: a single-chapter book's citations are bare verses with no
+ chapter at all, so a leading comma-piece that looks like a number is
+ always a VERSE, never a chapter introduction -- see book.ml's own
+ citation for the real, wrongly-parsed example this was found on. *)
+let parse_part ~single_chapter ~inherited s =
match split_on ':' s with
| [ c; v ] -> (
(* explicit "chapter:verses" *)
match (int_opt c, parse_ranges v) with
| Some ch, Some vs -> Some { chapter = ch; verses = vs }
| _ -> None)
- | [ only ] -> (
- (* Either "chapter, verses" (rule 3) or bare verses inheriting a
- chapter. Distinguish on whether the FIRST comma-piece is a lone
- number followed by more pieces -- a leading number followed by
- at least one further piece is a chapter introduction ("15, 1-46");
- a lone piece on its own, or a part with no more pieces to follow,
- can only be verses inheriting the previous chapter. *)
- let pieces = split_on ',' only in
- match (pieces, inherited) with
- | first :: (_ :: _ as rest), _ when int_opt first <> None && String.contains only ',' -> (
- (* "15, 1-46" -> chapter 15. This never fires for a part that
- already named its chapter via ":" -- those match the [c; v]
- branch above and never reach here. *)
- match (int_opt first, parse_ranges (String.concat "," rest)) with
- | Some ch, Some vs -> Some { chapter = ch; verses = vs }
- | _ -> None)
- | _, Some ch -> (
- match parse_ranges only with
- | Some vs -> Some { chapter = ch; verses = vs }
- | None -> None)
- | _, None -> None)
+ | [ only ] ->
+ if single_chapter then
+ match parse_ranges only with
+ | Some vs -> Some { chapter = Option.value inherited ~default:1; verses = vs }
+ | None -> None
+ else (
+ (* Either "chapter, verses" (rule 3) or bare verses inheriting a
+ chapter. Distinguish on whether the FIRST comma-piece is a lone
+ number followed by more pieces -- a leading number followed by
+ at least one further piece is a chapter introduction ("15, 1-46");
+ a lone piece on its own, or a part with no more pieces to follow,
+ can only be verses inheriting the previous chapter. *)
+ let pieces = split_on ',' only in
+ match (pieces, inherited) with
+ | first :: (_ :: _ as rest), _ when int_opt first <> None && String.contains only ',' -> (
+ (* "15, 1-46" -> chapter 15. This never fires for a part that
+ already named its chapter via ":" -- those match the [c; v]
+ branch above and never reach here. *)
+ match (int_opt first, parse_ranges (String.concat "," rest)) with
+ | Some ch, Some vs -> Some { chapter = ch; verses = vs }
+ | _ -> None)
+ | _, Some ch -> (
+ match parse_ranges only with
+ | Some vs -> Some { chapter = ch; verses = vs }
+ | None -> None)
+ | _, None -> None)
| _ -> None
let parse s =
@@ -144,10 +190,11 @@ let parse s =
(* A trailing ";" leaves an empty piece: drop it rather than
failing. Only if nothing remains is it an error. *)
let pieces = List.filter (fun p -> p <> "") (split_on ';' tail) in
+ let single_chapter = Book.is_single_chapter book in
let rec go inherited = function
| [] -> Ok []
| p :: rest -> (
- match parse_part ~inherited p with
+ match parse_part ~single_chapter ~inherited p with
| None -> Error ("cannot read reference: " ^ p)
| Some part -> (
match go (Some part.chapter) rest with
diff --git a/lib/citation/parse.mli b/lib/citation/parse.mli
index 3d1053c..ea38d24 100644
--- a/lib/citation/parse.mli
+++ b/lib/citation/parse.mli
@@ -8,7 +8,14 @@
([John 18:1-40; 19:1-42]) and really does list disjoint verse ranges
within a chapter ([Dan 13:1-9, 15-17, 19-30, 33-62]). *)
-type verse_range = { first : int; last : int option }
+(** A verse number, plus an optional trailing sub-verse letter run ([""] for
+ the overwhelming majority -- see parse.ml's own citation): ["11a"] is
+ [{ n = 11; suffix = "a" }], preserved through parsing and rendering
+ rather than dropped, so a reader never loses precision the source text
+ actually carried. *)
+type verse_num = { n : int; suffix : string }
+
+type verse_range = { first : verse_num; last : verse_num option }
type part = { chapter : int; verses : verse_range list }
type t = { book : Book.id; parts : part list }
diff --git a/lib/citation/render.ml b/lib/citation/render.ml
index 79aaa86..2b40ab3 100644
--- a/lib/citation/render.ml
+++ b/lib/citation/render.ml
@@ -80,13 +80,11 @@ let subst tmpl pairs =
Buffer.contents b
let render st ~names (t : Parse.t) =
+ let show_num (v : Parse.verse_num) = string_of_int v.Parse.n ^ v.Parse.suffix in
let one_range (r : Parse.verse_range) =
match r.Parse.last with
- | None -> string_of_int r.Parse.first
- | Some l ->
- subst st.range
- [ ("first", string_of_int r.Parse.first);
- ("last", string_of_int l) ]
+ | None -> show_num r.Parse.first
+ | Some l -> subst st.range [ ("first", show_num r.Parse.first); ("last", show_num l) ]
in
let one_part (p : Parse.part) =
let verses = String.concat st.verse_sep (List.map one_range p.Parse.verses) in
diff --git a/test/test_citation.ml b/test/test_citation.ml
index dae4610..3cbc8c6 100644
--- a/test/test_citation.ml
+++ b/test/test_citation.ml
@@ -267,10 +267,11 @@ module P = Colitur_citation.Parse
(* Render a parse back to a debug string so a test can assert shape
compactly: "book|chapter:v-v,v-v|chapter:v". *)
let show (t : P.t) =
+ let num (v : P.verse_num) = string_of_int v.P.n ^ v.P.suffix in
let range (r : P.verse_range) =
match r.P.last with
- | None -> string_of_int r.P.first
- | Some l -> Printf.sprintf "%d-%d" r.P.first l
+ | None -> num r.P.first
+ | Some l -> Printf.sprintf "%s-%s" (num r.P.first) (num l)
in
let part (p : P.part) =
Printf.sprintf "%d:%s" p.P.chapter
@@ -396,7 +397,9 @@ let test_subst_unknown_placeholder_alone () =
the assertion isolates exactly the numeral and nothing else. *)
let roman_of n =
let luke = match B.of_token "Luke" with Some id -> id | None -> assert false in
- let t : P.t = { P.book = luke; parts = [ { P.chapter = n; verses = [ { P.first = 1; last = None } ] } ] } in
+ let t : P.t =
+ { P.book = luke; parts = [ { P.chapter = n; verses = [ { P.first = { P.n = 1; suffix = "" }; last = None } ] } ] }
+ in
let style = R.style_of_fields [ ("book_sep", ""); ("chapter_verse", "{chapter_roman}") ] in
R.render style ~names:(fun _ _ -> "") t
diff --git a/test/test_citation_coverage_of.ml b/test/test_citation_coverage_of.ml
new file mode 100644
index 0000000..3444706
--- /dev/null
+++ b/test/test_citation_coverage_of.ml
@@ -0,0 +1,140 @@
+(* SPDX-License-Identifier: AGPL-3.0-or-later *)
+
+(* Fix wave I2 (final-review.md, 2026-08-25-colitur-of-phases-3-5): the OF
+ counterpart of test_citation_coverage.ml. That file's own header states
+ the rule this one follows: "every citation the engine can emit must
+ parse". EF reaches zero unconverted citations; OF does not, and this
+ file is the honest record of exactly where it falls short, mirroring the
+ review's own explicit instruction -- "enumerate them rather than
+ lowering the bar to fit".
+
+ MEASURED, not estimated: before this fix wave, 259 of 730 citation
+ fields in `colitur readings --rite of 2026` (35%) printed unconverted --
+ 345 distinct book spellings unregistered ({!Colitur_citation.Book}),
+ plus verse sub-letter markers ("11a") and a handful of single-chapter-
+ book misreads ({!Colitur_citation.Book.is_single_chapter}) the parser
+ could not read at all. All of that is now fixed at the source (book.ml's
+ own [of_lectionary_table] and parse.ml's own [verse_num]/[single_chapter]
+ additions), closing every genuinely fixable shape. What remains, walking
+ the FULL data/of/lectionary.sexp (not merely what one civil year happens
+ to observe -- a stricter check than EF's own year-walk, and the same one
+ tools/bootstrap_lectionary_of.ml's own generated header now runs and
+ discloses): 41 distinct references, out of 1540 total, EVERY ONE a
+ hyphenated range crossing a chapter boundary ("2:29-3:6", "1 John") --
+ a {!Colitur_citation.Parse.t} shape [part]/[verse_range] do not
+ represent (a [part] is one chapter and a list of within-chapter ranges;
+ representing a cross-chapter range would need restructuring that type
+ and {!Colitur_citation.Render}'s own consumption of it), deliberately
+ not built this task.
+
+ This file pins that residual EXACTLY -- not merely "at least these fail"
+ -- so it fails loudly in BOTH directions: a NEW unconverted reference
+ appearing (a regression) and one of these 41 starting to convert (this
+ list going stale, e.g. because a future task builds cross-chapter
+ support) are both caught, the same discipline every allow-list in this
+ project's suite already follows (`expected_rows`, `expected_bands`, the
+ litcal/missalemeum comparators). *)
+
+let lectionary_path = "../data/of/lectionary.sexp"
+
+let lectionary () =
+ match Colitur_kernel.Lectionary.load lectionary_path with
+ | Ok l -> l
+ | Error e -> Alcotest.failf "%s: %s" lectionary_path e
+
+(* The exact, pinned residual -- copied from data/of/lectionary.sexp's own
+ generated "Citations that do not parse" listing, itself produced by the
+ SAME {!Colitur_citation.Parse.parse} call this test makes, so the two
+ can never silently drift apart without one of them visibly failing to
+ regenerate/re-test clean. *)
+let known_cross_chapter_residual =
+ [ "1 Corinthians 12:31-13:13"; "1 John 1:5-2:2"; "1 John 2:29-3:6"; "1 John 3:22-4:6";
+ "1 John 4:19-5:4"; "2 Corinthians 3:15-4:1,3-6"; "2 Samuel 18:9-10,14b,24-25a,30-19:3";
+ "Colossians 1:24-2:3"; "Ecclesiastes 11:9-12:8"; "Ephesians 4:32-5:8"; "Exodus 11:10-12:14";
+ "Exodus 14:21-15:1"; "Ezekiel 2:8-3:4"; "Galatians 4:22-24,26-27,31-5:1"; "Genesis 1:1-2:2";
+ "Genesis 1:20-2:4a"; "Habakkuk 1:12-2:4"; "Hebrews 7:25-8:6"; "Isaiah 52:13-53:12";
+ "Isaiah 8:23b-9:3"; "John 15:26-16:4a"; "John 18:1-19:42"; "Jonah 1:1-2:2,11"; "Luke 7:36-8:3";
+ "Malachi 1:14b-2:2b,8-10"; "Mark 2:23-3:6"; "Mark 8:34-9:1"; "Matthew 10:34-11:1";
+ "Matthew 18:21-19:1"; "Matthew 9:35-10:1,5a,6-8"; "Matthew 9:36-10:8";
+ "Numbers 13:1-2,25-14:1,26a-29a,34-35"; "Philippians 3:17-4:1"; "Revelation 20:1-4,11-21:2";
+ "Sirach 27:30-28:7"; "The Acts 12:24-13:5a"; "The Acts 17:15,22-18:1"; "The Acts 7:51-8:1a";
+ "Wisdom 11:22-12:2"; "Wisdom 2:23-3:9"; "Wisdom 7:22b-8:1" ]
+
+(* Every failure this walk finds really is a chapter-crossing hyphen range
+ (a "-" whose right side itself contains a ":") -- checked mechanically,
+ not merely asserted, so a future genuinely-different failure shape
+ cannot hide inside this test by coincidentally also being in the pinned
+ list above. No [Str]/regex (frozen deps): plain character scanning. *)
+let looks_chapter_crossing reference =
+ let n = String.length reference in
+ let rec scan i saw_dash =
+ if i >= n then false
+ else if reference.[i] = '-' then scan (i + 1) true
+ else if reference.[i] = ':' && saw_dash then true
+ else scan (i + 1) saw_dash
+ in
+ scan 0 false
+
+let test_of_citations_convert_or_are_the_known_cross_chapter_residual () =
+ let total = ref 0 in
+ let bad = ref [] in
+ List.iter
+ (fun (_slug, cits) ->
+ List.iter
+ (fun (c : Colitur_kernel.Citation.t) ->
+ incr total;
+ match Colitur_citation.Parse.parse c.Colitur_kernel.Citation.reference with
+ | Ok _ -> ()
+ | Error _ ->
+ if not (List.mem c.Colitur_kernel.Citation.reference !bad) then
+ bad := c.Colitur_kernel.Citation.reference :: !bad)
+ cits)
+ (Colitur_kernel.Lectionary.entries (lectionary ()));
+ let bad = List.sort compare !bad in
+ Alcotest.(check int) "1540 citation fields in data/of/lectionary.sexp today -- if this changes, the \
+ pinned residual below may need updating too, not just this count"
+ 1540 !total;
+ Alcotest.(check (list string)) "the unconverted set is EXACTLY the known cross-chapter residual, no \
+ more and no fewer"
+ (List.sort compare known_cross_chapter_residual) bad;
+ List.iter
+ (fun r ->
+ Alcotest.(check bool) (Printf.sprintf "%s: really is a chapter-crossing range, not some other \
+ unrelated failure shape" r)
+ true (looks_chapter_crossing r))
+ bad
+
+(* Everything OUTSIDE the pinned residual round-trips through Sigla, the
+ same structural (not textual) comparison test_citation_coverage.ml's own
+ EF version makes, and for the identical reason its own header gives. *)
+let test_round_trip_outside_the_residual () =
+ let sg =
+ Colitur_citation.Sigla.make ~style:Colitur_citation.Render.default_style
+ ~tradition:Colitur_citation.Book.vulgate
+ ~names:(fun id _form -> Colitur_citation.Book.default_spelling id)
+ in
+ let bad = ref [] in
+ List.iter
+ (fun (_slug, cits) ->
+ List.iter
+ (fun (c : Colitur_kernel.Citation.t) ->
+ let r = c.Colitur_kernel.Citation.reference in
+ if not (List.mem r known_cross_chapter_residual) then
+ match Colitur_citation.Parse.parse r with
+ | Error e -> bad := Printf.sprintf "%s: unexpectedly does not parse (%s)" r e :: !bad
+ | Ok first -> (
+ let rendered = Colitur_citation.Sigla.format sg r in
+ match Colitur_citation.Parse.parse rendered with
+ | Error e -> bad := Printf.sprintf "%s: re-parse failed: %s" r e :: !bad
+ | Ok again -> if again <> first then bad := Printf.sprintf "%s: structure changed: %s" r rendered :: !bad))
+ cits)
+ (Colitur_kernel.Lectionary.entries (lectionary ()));
+ Alcotest.(check (list string)) "every OF citation outside the pinned residual round-trips through Sigla"
+ [] (List.sort compare !bad)
+
+let suite =
+ ( "citation-coverage-of",
+ [ Alcotest.test_case "every OF citation converts, or is the pinned cross-chapter residual" `Slow
+ test_of_citations_convert_or_are_the_known_cross_chapter_residual;
+ Alcotest.test_case "every OF citation outside the residual round-trips" `Slow
+ test_round_trip_outside_the_residual ] )
diff --git a/test/test_colitur.ml b/test/test_colitur.ml
index 6c00ee4..a47576b 100644
--- a/test/test_colitur.ml
+++ b/test/test_colitur.ml
@@ -6,6 +6,7 @@ let () =
Test_lang.suite;
Test_lang_coverage.suite;
Test_citation_coverage.suite;
+ Test_citation_coverage_of.suite;
Test_citation.suite;
("parse", Test_citation.parse_suite);
Test_citation.render_suite;