aboutsummaryrefslogtreecommitdiff
path: root/test/test_citation.ml
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-08-20 19:03:02 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-08-20 19:03:02 +0200
commit1da70dc7ac03fe33fb92b172a0e26932764170d6 (patch)
tree01e1ee37a2be97a0b06fad70de975d8007beb702 /test/test_citation.ml
parent46ffa91fd4fcc2249cd097b3b3a639d37a7d592e (diff)
downloadcolitur-1da70dc7ac03fe33fb92b172a0e26932764170d6.tar.gz
colitur-1da70dc7ac03fe33fb92b172a0e26932764170d6.zip
fix(citation): close the final review's blocking findings
The branch was RED and reported green. `dune test` exited 1: test/cli.t pinned the pre-fix output `kings_1 19:3-8`, which the previous commit had already fixed to `1 Reg 19:3-8`. The gate command piped dune through `tail`, so it reported tail's exit status, and cram prints its diff BEFORE the alcotest summary, so the two lines shown were the passing ones. Verify with `dune test; echo $?`, never through a pipe. A style file's own `book` key was unreachable. sigla_book resolved against a hardcoded "abbr" and the result was applied unconditionally, so the documented `[sigla] book = full` could never win. Render gains book_string, and the style's own value is now the default that a flag or config overrides. The unit test pinned style_of_fields correctly while the wiring defeated it. `lang --check` filtered the reference set to the celebration prefix, so a file with no [bible] section at all reported a clean bill of health -- contradicting both the reason the keys change was made and lang.ml's own comment. It now reports missing book names too. The token test missed a FOURTH citation-bearing file: adjustments.sexp writes citations as `Set_citation`, not `(reference ...)`. Its 16 citations all parse, so nothing was broken, but nothing was checking. The first attempt at this fix read the file and extracted NOTHING -- the marker stopped before the opening quote, so every payload was the part label -- which is recorded in the code rather than left as a trap. Also: colitur-config(5) claimed a trailing period the data does not carry, and two la.ini scan quotes silently corrected OCR damage ("Ionae 3, I - I O", "Epistolse") while presenting themselves as verbatim. Both are now marked as corrections.
Diffstat (limited to 'test/test_citation.ml')
-rw-r--r--test/test_citation.ml32
1 files changed, 26 insertions, 6 deletions
diff --git a/test/test_citation.ml b/test/test_citation.ml
index 1efaa10..133553b 100644
--- a/test/test_citation.ml
+++ b/test/test_citation.ml
@@ -117,8 +117,7 @@ let starts_with_at content pos prefix =
[Str]/regex -- a plain forward scan for the marker, then read to the
closing quote. Mirrors the coordinator's own survey command (grep -oh
over the reference marker, quote-delimited). *)
-let references content =
- let marker = "(reference \"" in
+let references_with marker content =
let mlen = String.length marker in
let len = String.length content in
let rec loop pos acc =
@@ -151,14 +150,35 @@ let book_token r =
let stop = if !i < len && r.[!i] = '.' then !i + 1 else !i in
String.sub r 0 stop
+(* A citation is written TWO ways in this project's data, and a survey that
+ knows only one of them silently under-reads the corpus:
+ (reference "Isa 60:1-6") -- lectionary/sanctoral/commons
+ (Set_citation First "Wis 7:7-14") -- adjustments.sexp, an overlay
+ Both markers are scanned below. *)
+let references content =
+ (* Each marker must include the OPENING QUOTE. [references_with] returns
+ the text between the marker and the next '"', so a marker stopping at
+ "(Set_citation " yields "First " / "Gospel " -- the part label, never
+ the citation -- and every entry is then silently discarded. That was
+ this function's first version, and it read the file while extracting
+ nothing at all. *)
+ references_with "(reference \"" content
+ @ references_with "(Set_citation First \"" content
+ @ references_with "(Set_citation Gospel \"" content
+
let test_every_data_file_token_resolves () =
(* The check whose absence caused fix round 1: the brief surveyed only
the lectionary and missed 21 tokens, several common, living in
- sanctoral.sexp and commons.sexp. Read all three files at test time and
- re-derive the token set from them, rather than hardcoding a list, so
- this keeps working when the data changes. *)
+ sanctoral.sexp and commons.sexp.
+
+ adjustments.sexp was then missed AGAIN, by this very test, because it
+ writes citations as `Set_citation` rather than `(reference ...)` -- the
+ same "one file too few" shape twice over. All FOUR shipped files are
+ read here, and the token set is re-derived from them at test time
+ rather than hardcoded, so this keeps working when the data changes. *)
let files =
- [ "../data/ef/lectionary.sexp"; "../data/ef/sanctoral.sexp"; "../data/ef/commons.sexp" ]
+ [ "../data/ef/lectionary.sexp"; "../data/ef/sanctoral.sexp";
+ "../data/ef/commons.sexp"; "../data/ef/adjustments.sexp" ]
in
let tokens =
files