aboutsummaryrefslogtreecommitdiff
path: root/lib/citation/parse.ml
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-08-26 23:53:49 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-08-26 23:53:58 +0200
commit90584d87e763789808329847c46dbaa8e22dca47 (patch)
tree6b5e7c29adc0d1cf5fdfb71cfb8f097867bf04b5 /lib/citation/parse.ml
parent9ebb06983b7a26db5564302253f2551dfbcf834e (diff)
downloadcolitur-90584d87e763789808329847c46dbaa8e22dca47.tar.gz
colitur-90584d87e763789808329847c46dbaa8e22dca47.zip
fix(citation): represent chapter-crossing verse ranges (W4)
Parse.verse_range's [last] endpoint gains an optional chapter (Parse.verse_end: { chapter : int option; verse : verse_num }), so a hyphen range whose two endpoints lie in different chapters ("1 John 1:5-2:2") can be represented at all. Rejected: a bare [int option] living alongside [last] as a second field on verse_range -- that would let 'a chapter with no verse' exist as a constructible value. parse_part now splits a part's leading "chapter:" at the FIRST colon only (not every colon), so the verses side can itself carry a second colon from a crossing range. parse_range detects a crossing by checking whether the range's right-hand side contains ':', and skips the same-chapter descending-range guard for that case (a later chapter is always "ahead", whatever its own verse numbers are). Handles the compound shape too -- a crossing range followed by further, same-chapter verse references in the same comma list ("Matthew 9:35-10:1,5a,6-8") -- since those trailing pieces parse as ordinary bare verses/ranges, unaffected by the preceding crossing. Render's one_range renders a crossing [last] through the same chapter_verse template one_part already uses for the part's own leading "chapter:verses", so a style that reconfigures the chapter/verse separator renders a crossing endpoint in that same convention rather than a hardcoded ':'. This closes the W4 known-wrong: 41 (now 49, after an intervening Second-reading extraction) of the OF lectionary's citations printed unconverted, every one this exact shape. test_citation_coverage_of.ml's pinned residual is now empty and asserted exactly, over the full 1725-field data/of/lectionary.sexp population, including the round-trip check (parse -> render -> parse structural equality). test_citation.ml gains direct parse-suite cases for the basic crossing, the compound shape, a mid-list crossing, a crossing with a sub-verse letter, and a malformed-crossing rejection. data/of/lectionary.sexp is regenerated via its own generator (tools/bootstrap_lectionary_of.ml, whose own embedded header text is updated to match); only comment lines change, confirmed by diff -- no lectionary entry differs. test_lectionary_of.ml's whole-file SHA-256 pin is updated to match. EF is unaffected: data/ef/ is untouched since v1.0.0, and a direct byte comparison of `colitur day`/`colitur readings` for 2026, 1583 and 9999 against a git-worktree build of 1c0137d is identical on all six outputs.
Diffstat (limited to 'lib/citation/parse.ml')
-rw-r--r--lib/citation/parse.ml76
1 files changed, 61 insertions, 15 deletions
diff --git a/lib/citation/parse.ml b/lib/citation/parse.ml
index 2c73600..811019e 100644
--- a/lib/citation/parse.ml
+++ b/lib/citation/parse.ml
@@ -13,12 +13,30 @@
it would silently lose real precision a reader can see in the source
text, trading "wrong format, complete" for "right format, incomplete". *)
type verse_num = { n : int; suffix : string }
-type verse_range = { first : verse_num; last : verse_num option }
+
+(* See parse.mli's own citation for [verse_end]'s rationale. *)
+type verse_end = { chapter : int option; verse : verse_num }
+type verse_range = { first : verse_num; last : verse_end option }
type part = { chapter : int; verses : verse_range list }
type t = { book : Book.id; parts : part list }
let split_on c s = String.split_on_char c s |> List.map String.trim
+(* Split at the FIRST ':' only, unlike [split_on ':'] which splits at every
+ one. [parse_part] needs this: a "chapter:verses" part whose verses
+ themselves contain a chapter-crossing hyphen range ("1:5-2:2") carries a
+ SECOND colon that belongs to the verses side, not to a second chapter
+ marker -- splitting on every ':' would see three pieces there and fail
+ to parse at all (the exact shape [parse_range]'s own citation above
+ restores). *)
+let split_first_colon s =
+ match String.index_opt s ':' with
+ | None -> None
+ | Some i ->
+ Some
+ ( String.trim (String.sub s 0 i),
+ String.trim (String.sub s (i + 1) (String.length s - i - 1)) )
+
(* The book is the longest leading run of non-digit words, allowing one
leading ordinal ("1 Cor", "3 Kings"). Everything after it is the
reference tail. *)
@@ -109,18 +127,44 @@ let num_opt s : verse_num option =
| None -> None
| Some (digits, suffix) -> ( match int_opt digits with Some n -> Some { n; suffix } | None -> None)
-(* "20-32" -> {first=20; last=Some 32}; "21" -> {first=21; last=None} *)
+(* "20-32" -> {first=20; last=Some{chapter=None; verse=32}};
+ "21" -> {first=21; last=None};
+ "35-10:1" -> {first=35; last=Some{chapter=Some 10; verse=1}} -- the
+ RIGHT side of a hyphen range may itself carry an explicit "chapter:verse"
+ (a range whose hyphen crosses a chapter boundary, [verse_end]'s own
+ citation in the .mli). The LEFT side never does: it is always the
+ chapter already in effect at that point in the part (parse_part's own
+ [c], or the chapter a PRECEDING range in the same part most recently
+ crossed into -- an ambiguity the source text itself never resolves and
+ this function does not try to, since nothing downstream needs the first
+ side's chapter: {!Colitur_citation.Render} only ever prints a part's
+ OWN [chapter] once, at the front, from [parse_part]'s [c]). *)
let parse_range s =
match split_on '-' s with
| [ a ] -> ( match num_opt a with Some f -> Some { first = f; last = None } | None -> None)
| [ a; b ] -> (
- match (num_opt a, num_opt b) with
- (* A descending range ("1:20-10") is always a transcription error;
- accepting it would render back out as a citation nobody can follow.
- Compared on the NUMBER only -- "11a-11b" is a real, ascending
- sub-verse range even though nothing here orders letters. *)
- | Some f, Some l when l.n >= f.n -> Some { first = f; last = Some l }
- | _ -> None)
+ match num_opt a with
+ | None -> None
+ | Some f ->
+ if String.contains b ':' then
+ (* The right side names its own chapter explicitly: no verse-
+ number ordering constraint applies (a later chapter is
+ always "ahead", whatever its own verse numbers are). *)
+ match split_on ':' b with
+ | [ cs; vs ] -> (
+ match (int_opt cs, num_opt vs) with
+ | Some c, Some v -> Some { first = f; last = Some { chapter = Some c; verse = v } }
+ | _ -> None)
+ | _ -> None
+ else
+ match num_opt b with
+ (* A descending range ("1:20-10") is always a transcription
+ error; accepting it would render back out as a citation
+ nobody can follow. Compared on the NUMBER only -- "11a-11b"
+ is a real, ascending sub-verse range even though nothing
+ here orders letters. *)
+ | Some l when l.n >= f.n -> Some { first = f; last = Some { chapter = None; verse = l } }
+ | _ -> None)
| _ -> None
let parse_ranges s =
@@ -140,13 +184,16 @@ let parse_ranges s =
always a VERSE, never a chapter introduction -- see book.ml's own
citation for the real, wrongly-parsed example this was found on. *)
let parse_part ~single_chapter ~inherited s =
- match split_on ':' s with
- | [ c; v ] -> (
- (* explicit "chapter:verses" *)
+ match split_first_colon s with
+ | Some (c, v) -> (
+ (* explicit "chapter:verses" -- [v] may itself carry further colons,
+ from a chapter-crossing hyphen range inside it; [parse_ranges] /
+ [parse_range] read those, this function does not need to. *)
match (int_opt c, parse_ranges v) with
| Some ch, Some vs -> Some { chapter = ch; verses = vs }
| _ -> None)
- | [ only ] ->
+ | None -> (
+ let only = s in
if single_chapter then
match parse_ranges only with
| Some vs -> Some { chapter = Option.value inherited ~default:1; verses = vs }
@@ -171,8 +218,7 @@ let parse_part ~single_chapter ~inherited s =
match parse_ranges only with
| Some vs -> Some { chapter = ch; verses = vs }
| None -> None)
- | _, None -> None)
- | _ -> None
+ | _, None -> None))
let parse s =
let s = String.trim s in