summaryrefslogtreecommitdiff
path: root/lib/citation
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-08-20 22:40:13 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-08-20 22:40:13 +0200
commitf73bd5d0e33146c24af1e98bf9ea26bd8cbf004b (patch)
tree790100dcc0077532d9446478e51d4cd40209c1c6 /lib/citation
parent2d8942e08edcbe5430f97270bdb1233e287600ee (diff)
parent671c4264707ec8c81845059746632b00f2481d88 (diff)
downloadcolitur-f73bd5d0e33146c24af1e98bf9ea26bd8cbf004b.tar.gz
colitur-f73bd5d0e33146c24af1e98bf9ea26bd8cbf004b.zip
Merge branch 'robustness-and-hackability'
An audit of the shipped program, plus the fixes it found. The citation parser accepted OCaml integer-literal syntax, so a typo like 'Luke 1_1:5' silently became a different chapter. Four pairs of different books shared a full title -- 1 and 2 Corinthians both rendered 'Epistola ad Corinthios' -- leaving 108 citations in 2027 alone that a reader could not resolve to a book. Spec section 8.5 is now delivered rather than recorded: shipped styles re-parse their own output. Overlay errors no longer name OCaml source files at the reader. Also: the new-overlay scaffold shows citations at the right nesting level, config --show validates before printing, error messages no longer echo whole file lines, and a month answers to both spellings of its own name.
Diffstat (limited to 'lib/citation')
-rw-r--r--lib/citation/book.ml116
-rw-r--r--lib/citation/parse.ml54
2 files changed, 119 insertions, 51 deletions
diff --git a/lib/citation/book.ml b/lib/citation/book.ml
index 2a2b9c0..986a898 100644
--- a/lib/citation/book.ml
+++ b/lib/citation/book.ml
@@ -12,57 +12,57 @@ let to_string t = t
the lectionary and was the source of every token missed in the first
pass. *)
let table =
- [ ("genesis", [ "Gen" ]);
- ("exodus", [ "Ex"; "Exod" ]);
- ("leviticus", [ "Lev" ]);
- ("numbers", [ "Num" ]);
- ("kings_3", [ "3 Kings"; "3 Kgs." ]);
- ("kings_4", [ "4 Kings" ]);
- ("esdras_2", [ "2 Esd." ]);
- ("tobit", [ "Tob" ]);
- ("judith", [ "Judith" ]);
- ("esther", [ "Esther" ]);
- ("proverbs", [ "Prov" ]);
- ("song_of_songs", [ "Song" ]);
- ("wisdom", [ "Wis"; "Wis." ]);
- ("ecclesiasticus", [ "Ecclus"; "Sir"; "Eccli" ]);
- ("isaiah", [ "Isa"; "Isa." ]);
- ("jeremiah", [ "Jer" ]);
- ("ezekiel", [ "Ezech"; "Ezek" ]);
- ("daniel", [ "Dan" ]);
- ("osee", [ "Osee" ]);
- ("joel", [ "Joel" ]);
- ("jonas", [ "Jonas" ]);
- ("malachi", [ "Mal" ]);
- ("matthew", [ "Matt"; "Matt." ]);
- ("mark", [ "Mark" ]);
- ("luke", [ "Luke" ]);
- ("john", [ "John" ]);
- ("acts", [ "Acts" ]);
- ("romans", [ "Rom" ]);
- ("corinthians_1", [ "1 Cor"; "1 Cor." ]);
- ("corinthians_2", [ "2 Cor"; "2 Cor." ]);
- ("galatians", [ "Gal" ]);
- ("ephesians", [ "Eph"; "Eph." ]);
- ("philippians", [ "Phil" ]);
- ("colossians", [ "Col"; "Col." ]);
- ("thessalonians_1", [ "1 Thess"; "1 Thess." ]);
- ("thessalonians_2", [ "2 Thess" ]);
- ("timothy_1", [ "1 Tim." ]);
- ("timothy_2", [ "2 Tim"; "2 Tim." ]);
- ("titus", [ "Titus" ]);
- ("hebrews", [ "Heb" ]);
- ("james", [ "Jas"; "James" ]);
- ("peter_1", [ "1 Pet"; "1 Pet." ]);
- ("peter_2", [ "2 Pet." ]);
- ("john_1", [ "1 John" ]);
+ [ ("genesis", [ "Gen"; "Liber Genesis"; "Genesis" ]);
+ ("exodus", [ "Ex"; "Exod"; "Liber Exodi"; "Exodus" ]);
+ ("leviticus", [ "Lev"; "Levit"; "Liber Levitici"; "Leviticus" ]);
+ ("numbers", [ "Num"; "Liber Numeri"; "Numbers" ]);
+ ("kings_3", [ "3 Kings"; "3 Kgs."; "3 Reg"; "Liber Regum III" ]);
+ ("kings_4", [ "4 Kings"; "4 Reg"; "Liber Regum IV"; "4 Kgs." ]);
+ ("esdras_2", [ "2 Esd."; "2 Esdr"; "Liber Esdrae"; "2 Esdras" ]);
+ ("tobit", [ "Tob"; "Liber Tobiae"; "Tobias" ]);
+ ("judith", [ "Judith"; "Iudith"; "Liber Iudith"; "Jth" ]);
+ ("esther", [ "Esther"; "Esth"; "Liber Esther" ]);
+ ("proverbs", [ "Prov"; "Proverbs" ]);
+ ("song_of_songs", [ "Song"; "Cant."; "Canticle of Canticles" ]);
+ ("wisdom", [ "Wis"; "Wis."; "Sap"; "Liber Sapientiae"; "Wisdom" ]);
+ ("ecclesiasticus", [ "Ecclus"; "Sir"; "Eccli"; "Ecclesiasticus" ]);
+ ("isaiah", [ "Isa"; "Isa."; "Isai"; "Isaias Propheta"; "Isaias" ]);
+ ("jeremiah", [ "Jer"; "Ier"; "Ieremias Propheta"; "Jeremias" ]);
+ ("ezekiel", [ "Ezech"; "Ezek"; "Ezechiel Propheta"; "Ezechiel" ]);
+ ("daniel", [ "Dan"; "Daniel Propheta"; "Daniel" ]);
+ ("osee", [ "Osee"; "Osee Propheta" ]);
+ ("joel", [ "Joel"; "Ioel"; "Ioel Propheta" ]);
+ ("jonas", [ "Jonas"; "Ionae"; "Ionas Propheta" ]);
+ ("malachi", [ "Mal"; "Malach"; "Malachias Propheta"; "Malachias" ]);
+ ("matthew", [ "Matt"; "Matt."; "Matth"; "Evangelium secundum Matthaeum"; "Matthew" ]);
+ ("mark", [ "Mark"; "Marc"; "Evangelium secundum Marcum" ]);
+ ("luke", [ "Luke"; "Luc"; "Evangelium secundum Lucam" ]);
+ ("john", [ "John"; "Ioann"; "Evangelium secundum Ioannem" ]);
+ ("acts", [ "Acts"; "Act"; "Actus Apostolorum"; "Acts of the Apostles" ]);
+ ("romans", [ "Rom"; "Epistola ad Romanos"; "Romans" ]);
+ ("corinthians_1", [ "1 Cor"; "1 Cor."; "Epistola I ad Corinthios"; "1 Corinthians" ]);
+ ("corinthians_2", [ "2 Cor"; "2 Cor."; "Epistola II ad Corinthios"; "2 Corinthians" ]);
+ ("galatians", [ "Gal"; "Epistola ad Galatas"; "Galatians" ]);
+ ("ephesians", [ "Eph"; "Eph."; "Ephes"; "Epistola ad Ephesios"; "Ephesians" ]);
+ ("philippians", [ "Phil"; "Epistola ad Philippenses"; "Philippians" ]);
+ ("colossians", [ "Col"; "Col."; "Epistola ad Colossenses"; "Colossians" ]);
+ ("thessalonians_1", [ "1 Thess"; "1 Thess."; "Epistola I ad Thessalonicenses"; "1 Thessalonians" ]);
+ ("thessalonians_2", [ "2 Thess"; "Epistola II ad Thessalonicenses"; "2 Thessalonians" ]);
+ ("timothy_1", [ "1 Tim."; "1 Tim"; "Epistola I ad Timotheum"; "1 Timothy" ]);
+ ("timothy_2", [ "2 Tim"; "2 Tim."; "Epistola II ad Timotheum"; "2 Timothy" ]);
+ ("titus", [ "Titus"; "Tit"; "Epistola ad Titum" ]);
+ ("hebrews", [ "Heb"; "Hebr"; "Epistola ad Hebraeos"; "Hebrews" ]);
+ ("james", [ "Jas"; "James"; "Iac"; "Epistola beati Iacobi Apostoli" ]);
+ ("peter_1", [ "1 Pet"; "1 Pet."; "1 Petri"; "Epistola I beati Petri Apostoli"; "1 Peter" ]);
+ ("peter_2", [ "2 Pet."; "2 Petri"; "Epistola II beati Petri Apostoli"; "2 Pet"; "2 Peter" ]);
+ ("john_1", [ "1 John"; "1 Ioann"; "Epistola beati Ioannis Apostoli" ]);
(* "Apoc" is the Vulgate spelling and "Rev" its modern equivalent, but
BOTH sit inside Vulgate-tradition data, so both resolve to the same
Vulgate id here -- see book.mli's note by [apocalypse] never being an
[of_token] result under that name. Do not add "revelation" as a
spelling: it exists only as a tradition target (below), and giving it
an [of_token] entry would let one book carry two different ids. *)
- ("apocalypse", [ "Apoc"; "Rev" ]) ]
+ ("apocalypse", [ "Apoc"; "Rev"; "Liber Apocalypsis"; "Apocalypse" ]) ]
(* Targets a tradition can map ONTO that the Vulgate data never cites
directly. Present so [tradition_of_fields] can validate both sides.
@@ -70,16 +70,34 @@ let table =
"Sir" and "Rev" already resolve to the Vulgate ids [ecclesiasticus] and
[apocalypse] above, so a modern-numbering tradition maps ONTO these
targets rather than data ever citing them directly. *)
-let tradition_targets =
- [ "kings_1"; "kings_2"; "nehemiah"; "sirach"; "hosea"; "jonah"; "revelation" ]
+
+(* The modern-numbering targets. They have table rows so that colitur's OWN
+ rendered output re-parses: with --sigla-tradition modern the book prints
+ as "1 Reg" or "1 Kings", and a user copying that back into an overlay must
+ get it read correctly. The shipped data never cites them directly, which
+ is why they are listed apart. *)
+let tradition_target_table =
+ [
+ ("kings_1", [ "1 Reg"; "Liber Regum I"; "1 Kgs"; "1 Kings" ]);
+ ("kings_2", [ "2 Reg"; "Liber Regum II"; "2 Kgs"; "2 Kings" ]);
+ ("nehemiah", [ "Neh"; "Liber Nehemiae"; "Nehemiah" ]);
+ ("sirach", [ "Liber Ecclesiastici"; "Sirach" ]);
+ ("hosea", [ "Os"; "Hos"; "Hosea" ]);
+ ("jonah", [ "Ion"; "Jon"; "Jonah" ]);
+ ("revelation", [ "Apocalypsis"; "Revelation" ]);
+ ]
+
+let tradition_targets = List.map fst tradition_target_table
let all = List.map fst table @ tradition_targets
let tokens =
- List.concat_map (fun (id, sp) -> List.map (fun s -> (s, id)) sp) table
+ List.concat_map
+ (fun (id, sp) -> List.map (fun s -> (s, id)) sp)
+ (table @ tradition_target_table)
let default_spelling id =
- match List.assoc_opt id table with
+ match List.assoc_opt id (table @ tradition_target_table) with
| Some (first :: _) -> first
| Some [] | None -> id
diff --git a/lib/citation/parse.ml b/lib/citation/parse.ml
index 45fbb72..2f2316e 100644
--- a/lib/citation/parse.ml
+++ b/lib/citation/parse.ml
@@ -9,7 +9,41 @@ let split_on c s = String.split_on_char c s |> List.map String.trim
(* The book is the longest leading run of non-digit words, allowing one
leading ordinal ("1 Cor", "3 Kings"). Everything after it is the
reference tail. *)
+(* Longest REGISTERED token that prefixes [s] and is followed by a space and
+ a digit. This is what lets a MULTI-WORD title parse: the heuristic below
+ stops at the first space, so "Evangelium secundum Lucam 5:12-14" would
+ otherwise split as the book "Evangelium" and fail.
+
+ Longest-match matters and is not decoration: "Liber Regum III" and
+ "Liber Regum IV" share a prefix with each other, and a shortest-match
+ would read both as some other book entirely. *)
+let longest_token_prefix s =
+ let n = String.length s in
+ let best = ref None in
+ List.iter
+ (fun (tok, _) ->
+ let tl = String.length tok in
+ if
+ tl < n
+ && String.sub s 0 tl = tok
+ && s.[tl] = ' '
+ (* a digit must follow, or "Job" would swallow the start of a
+ different book whose name merely begins the same way *)
+ && (let j = ref (tl + 1) in
+ while !j < n && s.[!j] = ' ' do incr j done;
+ !j < n && s.[!j] >= '0' && s.[!j] <= '9')
+ then
+ match !best with
+ | Some (b, _) when String.length b >= tl -> ()
+ | _ -> best := Some (tok, String.trim (String.sub s tl (n - tl)))
+ )
+ Book.tokens;
+ !best
+
let split_book s =
+ match longest_token_prefix s with
+ | Some (book, tail) when tail <> "" -> Some (book, tail)
+ | _ ->
let n = String.length s in
let i = ref 0 in
(* optional leading ordinal digit *)
@@ -25,7 +59,21 @@ let split_book s =
let tail = String.trim (String.sub s !i (n - !i)) in
if book = "" || tail = "" then None else Some (book, tail)
-let int_opt s = int_of_string_opt (String.trim s)
+(* A citation number is PLAIN DIGITS and positive -- nothing else.
+ [int_of_string_opt] also accepts OCaml's own integer-literal syntax, so
+ "1_1" would read as 11 and "+5" as 5: a transcription typo silently
+ becoming a DIFFERENT chapter, which nothing downstream could detect. A
+ user overlay supplies arbitrary strings, so this is reachable, not
+ theoretical. Chapter and verse numbering both start at 1, so zero is
+ rejected too. *)
+let int_opt s =
+ let s = String.trim s in
+ let ok =
+ s <> ""
+ && String.for_all (function '0' .. '9' -> true | _ -> false) s
+ in
+ if not ok then None
+ else match int_of_string_opt s with Some n when n > 0 -> Some n | _ -> None
(* "20-32" -> {first=20; last=Some 32}; "21" -> {first=21; last=None} *)
let parse_range s =
@@ -33,7 +81,9 @@ let parse_range s =
| [ a ] -> ( match int_opt a with Some f -> Some { first = f; last = None } | None -> None)
| [ a; b ] -> (
match (int_opt a, int_opt b) with
- | Some f, Some l -> Some { first = f; last = Some l }
+ (* A descending range ("1:20-10") is always a transcription error;
+ accepting it would render back out as a citation nobody can follow. *)
+ | Some f, Some l when l >= f -> Some { first = f; last = Some l }
| _ -> None)
| _ -> None