aboutsummaryrefslogtreecommitdiff
path: root/tools/extract_lms_ordo.ml
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-08-24 16:29:23 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-08-24 16:29:23 +0200
commit8a5da8756cf6b57aa36c01b71ce893f7967d3a29 (patch)
tree10873a692400fa25ce54c67ac9fc34d46af86272 /tools/extract_lms_ordo.ml
parent4f87cbde4ee79ac96aa7ebfed105b3a5ec02be43 (diff)
parent73b15551804bb63ee0081005e2869d36afb54be2 (diff)
downloadcolitur-8a5da8756cf6b57aa36c01b71ce893f7967d3a29.tar.gz
colitur-8a5da8756cf6b57aa36c01b71ce893f7967d3a29.zip
merge: celebrant Mass rubrics, and four rubrical corrections
colitur now prints what a real ordo prints for the Mass: which formulary is said and how it was reached, the Gloria, the Creed, the preface, and the commemorations with their Low-Mass/sung distinction. colitur rubrics joins day and readings. Four defects were found and fixed on the way, each by an external witness rather than by inspection: the Creed said at Requiem Masses (RG 476(f)), the missing bissextile shift of St Matthias and St Gabriel (2 041 leap years), Rogation Monday and Tuesday coloured violet in Paschaltide (RG 119(b)), and the ferias after the Ascension resuming the wrong Sunday's Mass rather than the Ascension's. Validation gained a sixth layer and then some: the preface is checked against three independent publishers over seven witness-years (FIUV, three LMS editions, three extraordinaryform.org editions), none of which shares the Divinum Officium -> missalemeum -> lectio lineage the older layers all descend from.
Diffstat (limited to 'tools/extract_lms_ordo.ml')
-rw-r--r--tools/extract_lms_ordo.ml729
1 files changed, 729 insertions, 0 deletions
diff --git a/tools/extract_lms_ordo.ml b/tools/extract_lms_ordo.ml
new file mode 100644
index 0000000..582fb6e
--- /dev/null
+++ b/tools/extract_lms_ordo.ml
@@ -0,0 +1,729 @@
+(* Task 6 (2026-08-21-colitur-celebrant-rubrics-phase1): turns pdftotext's
+ -layout dump of the Latin Mass Society's printed Ordo into
+ test/fixtures/lms-ordo-2024-2025.sexp, one row per UNIVERSAL-calendar
+ day (England & Wales diocesan variants are recorded as PRESENT via
+ [has_diocesan_variant] but their own text is discarded -- see the
+ fixture's own provenance header, and task-6-report.md's "how much of
+ the Ordo's year is comparable" accounting, for why).
+
+ NOT a general Ordo parser: hand-rolled against ONE PDF's own column
+ layout (Author "Peter Day-Milne", Producer "LibreOffice 7.4", pdftotext
+ 1.6), verified line-by-line against the source before being trusted --
+ see task-6-report.md for the full characterisation record (the
+ opposite-prediction probe technique CLAUDE.md's "burned twice" note
+ demands). Frozen deps forbid Str/regex; every match below is
+ String.sub/index/split_on_char, the same discipline tools/
+ bootstrap_sanctoral.ml's own hand-rolled INI reader already uses for
+ someone else's format.
+
+ THE KEY STRUCTURAL FACT this parser leans on, established during
+ characterisation: every day's UNIVERSAL entry (the General Roman
+ Calendar's own text) appears FIRST in reading order, before any
+ diocesan variant block (which is introduced by a line ending in ':',
+ e.g. "Westminster, Clifton, Plymouth:" or "In all Dioceses of ENGLAND
+ and WALES ... follows:"). So the FIRST "Gl"/"Cr" token pair and the
+ FIRST "Mass of ..."/roman-numeral-BVM line found within a day's block
+ are ALWAYS the universal entry's own, even if the colon-boundary
+ detection below is imperfect -- diocesan text can only ever appear
+ AFTER, never before, so it cannot introduce a false FIRST match. This
+ is why [scan_block] restricts its search to lines before the first
+ colon-terminated line (belt), while the ordering argument above is the
+ suspenders.
+
+ Usage:
+ pdftotext -layout docs/research/ordo/lms-ordo-2024-2025.pdf /tmp/lms.txt
+ dune exec tools/extract_lms_ordo.exe -- /tmp/lms.txt \
+ docs/research/ordo/lms-ordo-2024-2025.pdf \
+ test/fixtures/lms-ordo-2024-2025.sexp *)
+
+open Sexplib0.Sexp_conv
+module Date = Colitur_kernel.Date
+
+let die fmt = Printf.ksprintf (fun s -> prerr_endline ("extract_lms_ordo: " ^ s); exit 1) fmt
+
+(* Mirrored, not shared, by test/test_lms_ordo.ml -- tools/ and test/ have
+ no common .mli either could hang a shared type from, the same reasoning
+ test_oracle.ml's own header gives for duplicating [sha256_of_file]. *)
+type row = {
+ date : string; (** ISO-8601 *)
+ weekday : string;
+ (** As printed ("Sat", "Sun", ...) -- cross-checked below against
+ [date]'s own computed weekday at extraction time (fatal on
+ mismatch: the whole point of carrying both is to catch a date-
+ resolution bug here, not paper over one). *)
+ title : string; (** the day's own title line: feast name, class, colour, votive codes *)
+ formulary_override : string option;
+ (** the literal "Mass of ..." text, when the universal block names
+ one; [None] = the day says its own Mass. *)
+ bvm_numeral : string option;
+ (** "I".."V" when [formulary_override] is one of the five anchored
+ "<N> Mass of BVM" lines (whole-line match, never a substring --
+ see the module header on why "V Mass of BVM" must never be
+ matched as a substring of "IV Mass of BVM"). *)
+ gloria : bool option; (** [None] only if genuinely not found -- reported, not silently dropped *)
+ creed : bool option;
+ praef : string option;
+ (** RAW trailing text from the FIRST "Pr of ..." / "Common Pr" match
+ in the universal block's own reading order -- CAPTURED, NOT
+ CLASSIFIED here, the same discipline
+ tools/extract_fiuv_ordo.ml's own [row.praef] documents for a
+ different publisher's format. See the provenance header's own
+ "Pr of" characterisation bullet for what this column can and
+ cannot show: this Ordo prints OPTION LISTS ("Pr of X or Pr of Y
+ or Common Pr"), some of whose named options ("Martyrs", "All
+ Saints and Patron Saints", "the Dedication of a Church", "the
+ Most Holy Sacrament", "St John the Baptist", "the Angels") are
+ NOT among the fourteen the 1962 Missale Romanum's own RG 484-497
+ enumerate -- ad libitum extras this publisher includes alongside
+ a genuine RG 482 answer, never the sole option on any row found
+ in this corpus. Classification into {!Colitur_kernel.Preface.t}
+ happens test-side (test/test_lms_ordo.ml), by MEMBERSHIP in the
+ parsed option set, not string equality -- see that file's own
+ [classify_praef] citation for the full account. *)
+ has_diocesan_variant : bool;
+ (** whether a diocese-specific block followed the universal one
+ anywhere in this day's raw text (informational only -- see the
+ module header on why correctness of the other fields does not
+ depend on this being exact). *)
+}
+[@@deriving sexp]
+
+(* --- tiny hand-rolled helpers, no Str/regex ------------------------------ *)
+
+let split_ws s =
+ String.split_on_char ' ' (String.map (fun c -> if c = '\t' then ' ' else c) s)
+ |> List.filter (fun t -> t <> "")
+
+let is_all_digits s = s <> "" && String.for_all (fun c -> c >= '0' && c <= '9') s
+
+let has_prefix ~prefix s =
+ String.length s >= String.length prefix && String.sub s 0 (String.length prefix) = prefix
+
+let ends_with_colon s = s <> "" && s.[String.length s - 1] = ':'
+
+let months =
+ [| "January"; "February"; "March"; "April"; "May"; "June"; "July"; "August"; "September"; "October";
+ "November"; "December" |]
+
+let month_index name =
+ let rec go i = if i >= Array.length months then None else if months.(i) = name then Some (i + 1) else go (i + 1) in
+ go 0
+
+let weekday_of_abbrev = function
+ | "Sun" -> Some Date.Sun
+ | "Mon" -> Some Date.Mon
+ | "Tue" -> Some Date.Tue
+ | "Wed" -> Some Date.Wed
+ | "Thu" -> Some Date.Thu
+ | "Fri" -> Some Date.Fri
+ | "Sat" -> Some Date.Sat
+ | _ -> None
+
+(* Left-column codes (Abbreviations, "Left column": "+EW"/"+"/"Pl"/"Ind") --
+ printed in a dedicated margin column that pdftotext -layout reproduces
+ as literal leading tokens on an otherwise-ordinary content line. Strip
+ ALL that appear (there can be two, "Pl Ind" together -- the 31 Dec/
+ 1 Jan examples in the source). *)
+let strip_left_column tokens =
+ let is_marker = function "Pl" | "Ind" | "+EW" | "+" -> true | _ -> false in
+ let rec go = function t :: rest when is_marker t -> go rest | ts -> ts in
+ go tokens
+
+let bvm_numeral_strings = [ ("I", "I Mass of BVM"); ("II", "II Mass of BVM"); ("III", "III Mass of BVM");
+ ("IV", "IV Mass of BVM"); ("V", "V Mass of BVM") ]
+
+(* Anchored WHOLE-LINE match against all five, never a substring test --
+ this is the exact discipline the coordinator addendum records: "V Mass
+ of BVM" is a substring of "IV Mass of BVM", so any check here MUST
+ compare the full line, not search for the numeral's own string inside
+ it. *)
+let bvm_numeral_of_line line = List.assoc_opt line (List.map (fun (n, s) -> (s, n)) bvm_numeral_strings)
+
+(* --- the day-block scanner ------------------------------------------------ *)
+
+(* [lines]: this day's own raw continuation lines, marker-stripped and
+ trimmed, title line NOT included (the title never carries Gl/Cr/"Mass
+ of" itself in this source). Returns (formulary_override, bvm_numeral,
+ gloria, creed, praef, has_diocesan_variant), each derived from the
+ portion BEFORE the first colon-terminated (diocesan) line -- see the
+ module header for why this is belt-and-suspenders, not load-bearing on
+ its own. *)
+let scan_block lines =
+ let rec universal_prefix = function
+ | [] -> []
+ | l :: _ when ends_with_colon l -> []
+ | l :: rest -> l :: universal_prefix rest
+ in
+ let has_diocesan = List.exists ends_with_colon lines in
+ let universal = universal_prefix lines in
+ let formulary_override, bvm_numeral =
+ let rec find = function
+ | [] -> (None, None)
+ | l :: rest -> (
+ match bvm_numeral_of_line l with
+ | Some n -> (Some l, Some n)
+ | None -> if has_prefix ~prefix:"Mass of " l then (Some l, None) else find rest)
+ in
+ find universal
+ in
+ let all_tokens = List.concat_map split_ws universal in
+ let rec find_flag word = function
+ | prev :: (cur :: _ as rest) -> if cur = word then Some (prev <> "No") else find_flag word rest
+ | [ cur ] -> if cur = word then Some true else None
+ | [] -> None
+ in
+ (* [find_flag] needs a sentinel predecessor for a match at position 0
+ (never observed in this source -- Gl/Cr always follow at least "No"
+ or a preceding word on the same "Gl Cr Pr of ..." line -- but handled
+ rather than assumed: a bare leading "Gl"/"Cr" with no "No" before it
+ reads as [true], the same as any other non-"No" predecessor would). *)
+ let gloria = find_flag "Gl" ("" :: all_tokens) in
+ let creed = find_flag "Cr" ("" :: all_tokens) in
+ (* [praef]: LINE-SCOPED (not [all_tokens]'s flattened stream), because
+ the captured text runs to the END of whichever physical line carries
+ it -- flattening first would glue an unrelated following line's own
+ prose onto it. "Pr" always abbreviates "Preface" in this source's own
+ body text (checked directly: its only OTHER appearance is the front-
+ matter Abbreviations glossary entry "Pr Preface", outside the Ordo
+ body this tool scans), so a whole-token ("Pr","of") pair, or a bare
+ ("Common","Pr") pair when no proper/de-Tempore option is offered, is
+ the sole anchor needed -- no reject-list, unlike the BVM roman
+ numeral's own substring trap, because nothing else in this corpus's
+ body ever produces either two-token sequence. *)
+ let find_praef_in_line line =
+ let toks = Array.of_list (split_ws line) in
+ let n = Array.length toks in
+ let rec go i =
+ if i + 1 >= n then None
+ else if (toks.(i) = "Pr" && toks.(i + 1) = "of") || (toks.(i) = "Common" && toks.(i + 1) = "Pr") then
+ Some (String.concat " " (Array.to_list (Array.sub toks i (n - i))))
+ else go (i + 1)
+ in
+ go 0
+ in
+ let praef =
+ let rec find = function
+ | [] -> None
+ | l :: rest -> ( match find_praef_in_line l with Some s -> Some s | None -> find rest)
+ in
+ find universal
+ in
+ (formulary_override, bvm_numeral, gloria, creed, praef, has_diocesan)
+
+(* --- the top-level line scan --------------------------------------------- *)
+
+type block = { b_weekday : string; b_day : int; b_month : int; b_year : int; b_title : string; b_lines : string list }
+
+(* Generalised (Witnesses task, 2026-08-22) from a single hardcoded
+ sentinel ["End";"of";"November";"2024"], which assumed every edition
+ opens its Ordo body with an "End of November <year>" stub section. That
+ assumption FAILED characterisation against the two newly-acquired
+ editions: the 2023-2024 PDF has NO such stub -- Advent Sunday 2023 fell
+ on 3 December, so only 2 tail days of the old liturgical year remained
+ (1-2 December), and the compiler folded them directly into the
+ "December 2023" header rather than giving them their own mini-section.
+ So a start marker anchored to "End of November" cannot generalise.
+
+ What DOES generalise, checked directly against all three editions'
+ pdftotext -layout dumps: exactly one line in the whole document begins
+ with the bare, all-uppercase token "ORDO" -- either alone ("ORDO",
+ 2024-2025/2025-2026, introducing an "End of <Month> <Year>" stub next)
+ or fused with the edition's own year range ("ORDO 2023-2024",
+ introducing the first REAL month header directly, no stub). This is
+ NEVER the Table of Contents' own entry, which prints title-case "Ordo
+ 20XX-20XX" followed by dot leaders and a page number -- a different
+ string ("Ordo", not "ORDO"), confirmed by grepping all three raw dumps
+ for both patterns before relying on either. *)
+let is_ordo_marker tokens = match tokens with "ORDO" :: _ -> true | _ -> false
+
+(* The "End of <Month> <Year>" stub heading, when an edition has one --
+ parsed directly (never hardcoded to November/a specific year) so the
+ edition's own initial (month, year) state comes from the text itself,
+ not an assumption baked into the tool. *)
+let parse_end_of_month tokens =
+ match tokens with
+ | [ "End"; "of"; m; y ] when is_all_digits y -> ( match month_index m with Some mi -> Some (mi, int_of_string y) | None -> None)
+ | _ -> None
+
+let is_stop_tail_line tokens =
+ match tokens with tok :: _ -> String.length tok = 4 && is_all_digits tok | [] -> false
+
+let is_month_header tokens =
+ match tokens with
+ | m :: rest when month_index m <> None -> List.for_all is_all_digits rest
+ | _ -> false
+
+let read_lines path =
+ let ic = open_in path in
+ let rec loop acc = match input_line ic with l -> loop (l :: acc) | exception End_of_file -> List.rev acc in
+ let ls = loop [] in
+ close_in ic;
+ ls
+
+let () =
+ if Array.length Sys.argv <> 4 then
+ die "usage: extract_lms_ordo <pdftotext-layout.txt> <source.pdf> <dest.sexp>";
+ let txt_path = Sys.argv.(1) in
+ let pdf_path = Sys.argv.(2) in
+ let dest = Sys.argv.(3) in
+ let all_lines = read_lines txt_path in
+ let arr = Array.of_list all_lines in
+ let n = Array.length arr in
+ let trimmed i = String.trim arr.(i) in
+ let ordo_marker_idx =
+ let rec go i = if i >= n then die "no line begins with the bare \"ORDO\" body marker (see the tool's own comment on [is_ordo_marker] -- this edition's format may have drifted)"
+ else if is_ordo_marker (split_ws (trimmed i)) then i else go (i + 1)
+ in
+ go 0
+ in
+ (* The first non-blank line after the marker is either an "End of <Month>
+ <Year>" stub (consumed here as a pure sentinel, exactly as the old
+ hardcoded design did -- [start_idx] becomes ITS OWN index so the main
+ loop below begins right after it) or a real "<Month> <Year>" header
+ with an explicit year token (an edition with no stub, e.g. 2023-2024 --
+ [start_idx] becomes the ORDO marker's own index instead, so the main
+ loop re-encounters this header line on its very first iteration and
+ [is_month_header]'s own existing per-line handling sets [cur_month]/
+ [cur_year] from it normally). Either way [init_month]/[init_year] come
+ directly from parsed text, never a hardcoded constant -- a missing
+ explicit year on this FIRST header is a genuine format break, so it
+ dies loudly rather than silently seeding a wrong year. *)
+ let rec first_nonblank i =
+ if i >= n then die "no content found after the ORDO marker at line %d" ordo_marker_idx
+ else if trimmed i = "" then first_nonblank (i + 1)
+ else i
+ in
+ let content_idx = first_nonblank (ordo_marker_idx + 1) in
+ let content_toks = split_ws (trimmed content_idx) in
+ let start_idx, init_month, init_year =
+ match parse_end_of_month content_toks with
+ | Some (mi, y) -> (content_idx, mi, y)
+ | None -> (
+ match content_toks with
+ | m :: (y :: _ as rest) when month_index m <> None && List.for_all is_all_digits rest ->
+ (ordo_marker_idx, (match month_index m with Some x -> x | None -> assert false), int_of_string y)
+ | _ ->
+ die "first content after the ORDO marker (%S) is neither an \"End of <Month> <Year>\" stub nor a \"<Month> <Year>\" header carrying an explicit year"
+ (String.concat " " content_toks))
+ in
+ let stop_idx =
+ let rec go i =
+ if i >= n then n else if is_stop_tail_line (split_ws (trimmed i)) then i else go (i + 1)
+ in
+ go (start_idx + 1)
+ in
+ Printf.printf "extract_lms_ordo: Ordo body lines %d..%d (of %d total), initial month/year %d/%d\n" start_idx stop_idx n
+ init_month init_year;
+ (* --- the scan: month/year tracker + block accumulator ------------------ *)
+ let cur_month = ref init_month and cur_year = ref init_year in
+ let blocks = ref [] in
+ let cur = ref None in
+ let flush () = match !cur with Some b -> blocks := b :: !blocks; cur := None | None -> () in
+ for i = start_idx + 1 to stop_idx - 1 do
+ let raw = trimmed i in
+ if raw = "" then ()
+ else if String.length raw >= 1 && (let rec contains s sub =
+ let ls = String.length s and lu = String.length sub in
+ if lu = 0 then true else if ls < lu then false
+ else if String.sub s 0 lu = sub then true
+ else contains (String.sub s 1 (ls - 1)) sub
+ in
+ contains raw "Latin Mass Society Ordo")
+ then () (* page footer *)
+ else
+ let toks = split_ws raw in
+ if is_month_header toks then begin
+ match toks with
+ | m :: rest -> (
+ match month_index m with
+ | Some mi ->
+ if mi <> !cur_month then begin
+ if mi < !cur_month then incr cur_year;
+ cur_month := mi
+ end;
+ (* an explicit 4-digit year token, when present, overrides
+ the transition inference outright -- belt and
+ suspenders, same reasoning as [scan_block] above *)
+ List.iter (fun t -> if String.length t = 4 && is_all_digits t then cur_year := int_of_string t) rest
+ | None -> ())
+ | [] -> ()
+ end
+ else
+ match toks with
+ | wd :: dn :: _ when weekday_of_abbrev wd <> None && is_all_digits dn && int_of_string dn >= 1
+ && int_of_string dn <= 31 ->
+ flush ();
+ let title = String.concat " " (List.tl (List.tl toks)) in
+ cur := Some { b_weekday = wd; b_day = int_of_string dn; b_month = !cur_month; b_year = !cur_year;
+ b_title = title; b_lines = [] }
+ | _ -> (
+ match !cur with
+ | None -> ()
+ | Some b ->
+ let stripped = strip_left_column toks in
+ if stripped = [] then ()
+ else cur := Some { b with b_lines = b.b_lines @ [ String.concat " " stripped ] })
+ done;
+ flush ();
+ let blocks = List.rev !blocks in
+ Printf.printf "extract_lms_ordo: %d day-blocks parsed\n" (List.length blocks);
+ let rows =
+ List.map
+ (fun b ->
+ let date =
+ match Date.make ~year:b.b_year ~month:b.b_month ~day:b.b_day with
+ | Ok d -> d
+ | Error e -> die "%s %d/%d/%d: %s" b.b_weekday b.b_year b.b_month b.b_day e
+ in
+ (match weekday_of_abbrev b.b_weekday with
+ | Some w when w = Date.weekday date -> ()
+ | Some _ ->
+ die "%s: printed weekday %S does not match computed weekday %S -- date resolution bug"
+ (Date.to_iso8601 date) b.b_weekday (Date.weekday_to_string (Date.weekday date))
+ | None -> die "%s: unrecognised weekday abbreviation %S" (Date.to_iso8601 date) b.b_weekday);
+ let formulary_override, bvm_numeral, gloria, creed, praef, has_diocesan_variant = scan_block b.b_lines in
+ { date = Date.to_iso8601 date; weekday = b.b_weekday; title = b.b_title; formulary_override; bvm_numeral;
+ gloria; creed; praef; has_diocesan_variant })
+ blocks
+ in
+ let missing_gloria = List.filter (fun r -> r.gloria = None) rows in
+ let missing_creed = List.filter (fun r -> r.creed = None) rows in
+ let missing_praef = List.filter (fun r -> r.praef = None) rows in
+ if missing_gloria <> [] then
+ Printf.printf "extract_lms_ordo: WARNING %d rows with no Gloria found: %s\n" (List.length missing_gloria)
+ (String.concat ", " (List.map (fun r -> r.date) missing_gloria));
+ if missing_creed <> [] then
+ Printf.printf "extract_lms_ordo: WARNING %d rows with no Creed found: %s\n" (List.length missing_creed)
+ (String.concat ", " (List.map (fun r -> r.date) missing_creed));
+ if missing_praef <> [] then
+ Printf.printf "extract_lms_ordo: WARNING %d rows with no Pr-of/Common-Pr found: %s\n" (List.length missing_praef)
+ (String.concat ", " (List.map (fun r -> r.date) missing_praef));
+ let with_variant = List.length (List.filter (fun r -> r.has_diocesan_variant) rows) in
+ let with_override = List.length (List.filter (fun r -> r.formulary_override <> None) rows) in
+ let with_bvm = List.length (List.filter (fun r -> r.bvm_numeral <> None) rows) in
+ Printf.printf "extract_lms_ordo: %d rows; %d with a diocesan variant present; %d with a formulary override; \
+ %d with a BVM roman numeral; %d with no Pr-of/Common-Pr text\n"
+ (List.length rows) with_variant with_override with_bvm (List.length missing_praef);
+ (* --- provenance header + write ----------------------------------------- *)
+ let sha256_of_file path =
+ let cmd = Printf.sprintf "sha256sum %s" (Filename.quote path) in
+ let ic = Unix.open_process_in cmd in
+ let line = try input_line ic with End_of_file -> die "sha256sum produced no output for %s" path in
+ (match Unix.close_process_in ic with Unix.WEXITED 0 -> () | _ -> die "sha256sum failed for %s" path);
+ match String.index_opt line ' ' with Some i -> String.sub line 0 i | None -> die "unexpected sha256sum output: %S" line
+ in
+ let today () =
+ let tm = Unix.gmtime (Unix.time ()) in
+ Printf.sprintf "%04d-%02d-%02d" (tm.Unix.tm_year + 1900) (tm.Unix.tm_mon + 1) tm.Unix.tm_mday
+ in
+ let pdf_sha = sha256_of_file pdf_path in
+ (* --- per-edition provenance + characterisation, Witnesses task ---------
+ Generalised 2026-08-22 from Task 6's single hand-written header (which
+ named "lms-ordo-2024-2025.sexp" literally throughout). Facts 1/3/4
+ below are STRUCTURAL -- true of the publisher's whole known corpus,
+ re-checked directly against the two new editions' own pdftotext dumps
+ before being asserted here again, not merely copied forward. Fact 2
+ (the Creed opposite-prediction probes) is inherently edition-specific
+ (the actual calendar dates differ year to year, and one edition's own
+ calendar even breaks the standard pairing -- see the 2025-2026 note
+ below), so it is looked up per [dest] and, critically, RE-VERIFIED
+ against this run's own [rows] before being printed: a probe that
+ fails is a [die], not a silently-wrong "confirmed" claim in a fixture
+ header nobody will re-check by eye. *)
+ let base = Filename.basename dest in
+ let pdf_base = Filename.basename pdf_path in
+ let edition_label, isbn, copyright_year, pdf_meta, url, url_note, window_open_desc =
+ match base with
+ | "lms-ordo-2023-2024.sexp" ->
+ ( "2023-2024",
+ "978-1-7392096-2-9",
+ "2023",
+ {|Author "Peter Day-Milne", Producer "LibreOffice 6.4", CreationDate 2023-11-01, 140 pages|},
+ "https://lms.org.uk/sites/default/files/u5374/ordo_2023-2024_2.2_final.pdf",
+ "identified 2026-08-22 via web search matching the PDF's own edition year to the publisher's own hosting path; NOT independently re-downloaded and byte-compared in this session -- recorded honestly rather than presented as verified",
+ {|1 December 2023 (this edition carries NO "End of November" stub section at all -- Advent Sunday 2023 fell on 3 December, so only 2 tail days of the prior liturgical year remained, and the compiler folded them directly into the "December 2023" header instead of giving them their own mini-section; see CHARACTERISATION below)|}
+ )
+ | "lms-ordo-2025-2026.sexp" ->
+ ( "2025-2026",
+ "978-1-7392096-5-0",
+ "2025",
+ {|Author "Peter Day-Milne", Producer "LibreOffice 25.2.6.2 (X86_64) / LibreOffice Community", CreationDate 2025-10-23, 146 pages|},
+ "(not found -- see the note on this line)",
+ "NOT found by URL-pattern web search this session (unlike the 2023-2024 and 2024-2025 editions, no exact download link matching this PDF's own filename/version turned up under lms.org.uk/sites/default/files/u5374/; several near-miss candidates were checked and rejected -- none matched this PDF's own metadata) -- left honestly blank rather than guessed; the file was supplied locally to this session, not fetched by this tool",
+ {|28 November 2025 ("End of November 2025", the same shape as the 2024-2025 edition -- Advent Sunday 2025 fell on 30 November, so only 3 tail days remained; note this excludes 27 November 2025, which the PRECEDING edition's own window (2024-2025, through 2025-12-31) already covers -- a one-day-short overlap between consecutive editions' own windows, not a gap or a bug in either extraction|}
+ )
+ | "lms-ordo-2024-2025.sexp" ->
+ (* Preface-witnesses task (2026-08-23): brought into this same
+ auto-generated, probe-verified header template so all three
+ editions carry the identical characterisation discipline --
+ Task 6's own original HAND-WRITTEN header (never regenerated by
+ the Witnesses task, which only added table entries for the two
+ NEW editions) is superseded by this run; every fact below is
+ re-derived from this PDF's own text/metadata, not copied
+ blind from that older header (though it agrees with it in
+ every particular checked). *)
+ ( "2024-2025",
+ "9781739209636",
+ "2024",
+ {|Author "Peter Day-Milne", Producer "LibreOffice 7.4", CreationDate 2024-11-06, 140 pages|},
+ "https://lms.org.uk/sites/default/files/u5374/ordo_2.21_2024-2025.pdf",
+ "identified via web search matching the PDF's own version number and filename to the publisher's own hosting path (product page: https://lms.org.uk/product/latin-mass-society-ordo-2024-25); NOT independently re-downloaded and byte-compared in this session -- recorded honestly rather than presented as verified",
+ {|27 November 2024 ("End of November 2024", the same shape as the 2025-2026 edition -- Advent Sunday 2024 fell on 1 December, so 4 tail days remained (27-30 November))|}
+ )
+ | _ -> die "no provenance/characterisation entry for %s -- add one to extract_lms_ordo.ml's own per-edition table before generating this fixture (the tool refuses to emit an under-characterised header)" base
+ in
+ let creed_of_date rows d = List.find_map (fun r -> if String.equal r.date d then r.creed else None) rows in
+ let probes =
+ match base with
+ | "lms-ordo-2023-2024.sexp" ->
+ [ ("2023-12-03", true, "Advent Sunday, RG 475(a)");
+ ("2023-12-04", false, "St Peter Chrysologus, III class, ordinary Advent day, RG 476(b)/(d)");
+ ("2023-12-26", true, "St Stephen, II class, inside the Nativity octave, RG 475(d)");
+ ("2024-08-15", true, "the Assumption, I class, RG 475(b)");
+ ("2024-08-16", false, "St Joachim, II class, no Lord/BVM/apostle clause applies, RG 476(b)") ]
+ | "lms-ordo-2025-2026.sexp" ->
+ [ ("2025-11-30", true, "Advent Sunday, RG 475(a)");
+ ("2025-12-01", false, "ordinary Advent feria, RG 476(b)/(d)");
+ ("2025-12-26", true, "St Stephen, II class, inside the Nativity octave, RG 475(d)");
+ ( "2026-08-10", false,
+ "St Laurence, II class martyr, no Lord/BVM/apostle clause applies, RG 476(b) -- St Joachim (16 August) is IMPEDED by a Sunday in this edition's own calendar (12th Sunday after Pentecost), so this edition needed a DIFFERENT sharp non-octave II-class pairing than the 2023-2024/2024-2025 editions used; a real, edition-specific finding, not an extraction quirk"
+ );
+ ("2026-08-15", true, "the Assumption, I class, RG 475(b)") ]
+ | "lms-ordo-2024-2025.sexp" ->
+ [ ("2024-12-01", true, "Advent Sunday, RG 475(a)");
+ ("2024-12-02", false, "ordinary Advent feria, RG 476(b)/(d)");
+ ("2024-12-26", true, "St Stephen, II class, inside the Nativity octave, RG 475(d)");
+ ("2025-08-15", true, "the Assumption, I class, RG 475(b)");
+ ("2025-08-16", false, "St Joachim, II class, no Lord/BVM/apostle clause applies, RG 476(b)") ]
+ | _ -> die "no probe list for %s" base
+ in
+ List.iter
+ (fun (d, expected, desc) ->
+ match creed_of_date rows d with
+ | Some got when Bool.equal got expected -> ()
+ | Some got -> die "characterisation probe FAILED: %s (%s) -- expected creed=%b, extracted creed=%b" d desc expected got
+ | None -> die "characterisation probe FAILED: %s (%s) -- no row extracted for this date at all" d desc)
+ probes;
+ Printf.printf "extract_lms_ordo: all %d characterisation probes confirmed for %s\n" (List.length probes) edition_label;
+ (* Preface-witnesses task (2026-08-23): a SECOND opposite-prediction
+ probe set, over [praef] rather than [creed] -- the same discipline,
+ applied to the new column this task adds, per the task brief's own
+ "probe dates where the rule predicts OPPOSITE answers before
+ adjudicating anything" instruction. Three universal, unambiguous
+ fixed-date feasts per edition, each landing on a DIFFERENT named
+ preface (RG 484 Nativity, RG 497 Apostles, RG 499 Requiem) -- checked
+ directly against this run's own [rows] before this header is
+ written, a hard [die] on failure, not a printed claim. All Souls'
+ own date is edition-specific (2 November, TRANSFERRED to 3 November
+ in the one edition -- 2024-2025 -- where 2 November falls on a
+ Sunday; verified directly against each PDF's own text, not assumed
+ from RG 96 alone). *)
+ let praef_of_date rows d = List.find_map (fun r -> if String.equal r.date d then r.praef else None) rows in
+ let praef_probes =
+ match base with
+ | "lms-ordo-2023-2024.sexp" ->
+ [ ("2023-12-25", "Pr of the Nativity", "the Nativity of Our Lord, I class, RG 484(a)");
+ ("2024-06-29", "Pr of the Apostles", "SS Peter & Paul, I class, RG 497");
+ ("2024-11-02", "Pr of the Dead", "All Souls' Day, I class, RG 499") ]
+ | "lms-ordo-2024-2025.sexp" ->
+ [ ("2024-12-25", "Pr of the Nativity", "the Nativity of Our Lord, I class, RG 484(a)");
+ ("2025-06-29", "Pr of the Apostles", "SS Peter & Paul, I class, RG 497");
+ ( "2025-11-03", "Pr of the Dead",
+ "All Souls' Day, I class, RG 499 -- TRANSFERRED from 2 November, a Sunday in this edition's own calendar" )
+ ]
+ | "lms-ordo-2025-2026.sexp" ->
+ [ ("2025-12-25", "Pr of the Nativity", "the Nativity of Our Lord, I class, RG 484(a)");
+ ("2026-06-29", "Pr of the Apostles", "SS Peter & Paul, I class, RG 497");
+ ("2026-11-02", "Pr of the Dead", "All Souls' Day, I class, RG 499") ]
+ | _ -> die "no praef probe list for %s" base
+ in
+ List.iter
+ (fun (d, expected_prefix, desc) ->
+ match praef_of_date rows d with
+ | Some got when has_prefix ~prefix:expected_prefix got -> ()
+ | Some got ->
+ die "praef characterisation probe FAILED: %s (%s) -- expected prefix %S, extracted %S" d desc
+ expected_prefix got
+ | None -> die "praef characterisation probe FAILED: %s (%s) -- no praef extracted for this date at all" d desc)
+ praef_probes;
+ Printf.printf "extract_lms_ordo: all %d praef characterisation probes confirmed for %s\n" (List.length praef_probes)
+ edition_label;
+ let praef_probe_lines =
+ String.concat "\n"
+ (List.map (fun (d, expected_prefix, desc) -> Printf.sprintf "; %s (%s) -> %S" d desc expected_prefix) praef_probes)
+ in
+ let missing_praef_count = List.length missing_praef in
+ let probe_lines =
+ String.concat "\n"
+ (List.map
+ (fun (d, expected, desc) -> Printf.sprintf "; %s (%s) -> %s" d desc (if expected then "TRUE" else "FALSE"))
+ probes)
+ in
+ let header =
+ Printf.sprintf
+ {|; %s -- Witnesses task (2026-08-22-colitur-celebrant-rubrics-phase1),
+; extending Task 6's single LMS fixture (2024-2025) to three editions of
+; the Latin Mass Society's own printed Ordo for England & Wales -- a
+; lineage independent of Divinum Officium/missalemeum/lectio (layers 3-4)
+; and of the electronic LT.txt transcription (layers 1-2/5): a physically
+; printed, professionally-compiled liturgical calendar with its own ISBN.
+;
+; Generator: tools/extract_lms_ordo.ml -- do not hand-edit; re-run against
+; a fresh pdftotext dump and commit the diff instead.
+;
+; Source: "The Ordo %s", compiled by Peter Day-Milne, The Latin
+; Mass Society, 9 Mallow Street, London EC1Y 8RQ (title page and
+; colophon, verified directly against the PDF's own extracted text --
+; ISBN %s, Copyright (c) The Latin Mass Society, UK %s).
+; PDF metadata: %s.
+; URL (%s):
+; %s
+; The local PDF (docs/research/ordo/%s) is gitignored
+; (CLAUDE.md: "docs/ is gitignored"), so this header, not git, is this
+; fixture's only record of where it came from.
+; SHA-256 of docs/research/ordo/%s: %s
+; Extracted (UTC): %s
+; Exact commands:
+; pdftotext -layout docs/research/ordo/%s /tmp/lms.txt
+; dune exec tools/extract_lms_ordo.exe -- /tmp/lms.txt \
+; docs/research/ordo/%s %s
+;
+; COVERAGE: %d day-rows, %s through %s (starting %s,
+; through the day before a malformed trailing entry the extractor
+; deliberately stops before -- see CHARACTERISATION below).
+;
+; CHARACTERISATION FINDINGS (Step 1, non-negotiable per the task brief --
+; the extraordinaryform.org Ordo silently omitted St Lawrence's vigil on
+; EVERY date it covered, which nearly produced a false corroboration
+; during the RG 33 work; every LMS edition, including this one, is probed
+; with the same discipline before a single divergence is adjudicated):
+;
+; 1. Gl/Cr are printed PER MASS-OPTION, not per day (confirmed again for
+; THIS edition: it carries many more Gl/Cr pairs than day-rows, because
+; every diocesan variant repeats its own "Gl ... Cr ... Pr of ..."
+; line). This extractor resolves the day's OWN office by taking the
+; FIRST Gl/Cr pair in reading order, which the source's own layout
+; guarantees is the universal entry's (diocesan variants are always
+; introduced by a colon-terminated diocese-list line that comes AFTER
+; the universal block, never before -- checked structurally, not
+; assumed, in every sampled block during development, this edition
+; included).
+;
+; 2. OPPOSITE-PREDICTION PROBES against Rite_ef.Rubrics_ef.creed (RG
+; 475-476), RE-CHECKED BY THIS TOOL RUN (not merely copied from a prior
+; edition's findings) before a single row of this fixture is trusted --
+; the technique CLAUDE.md records as the one that catches a source
+; silently omitting the very thing it is meant to witness. This run's
+; own extracted [rows] were checked against every one of the following
+; before this header was even written; a failure here is a hard [die],
+; not a printed claim:
+%s
+; All %d predictions confirmed, both directions (TRUE and FALSE each
+; independently witnessed) -- this edition is discriminating, not a
+; constant, on the Creed.
+;
+; 3. The BVM-Saturday roman numeral (I-V, "Missae de sancta Maria in
+; sabbato", RG 309(a)) is anchored by WHOLE-LINE match against the five
+; literal strings below, NEVER by substring search: "V Mass of BVM" is
+; a substring of "IV Mass of BVM", so a naive `contains` check
+; misreads every fourth-Mass day as the fifth (the coordinator's own
+; misreading, recorded and retracted in commit 27b07b4 before Task 6
+; existed; the same anchoring logic runs unchanged for every edition).
+;
+; 4. SCOPE EXCLUDED, quantified, not silently dropped:
+; - Diocesan variants: present on %d of %d days (%s%%) -- their own
+; text is read only far enough to set [has_diocesan_variant],
+; never compared, because colitur computes the UNIVERSAL General
+; Roman Calendar only, with no diocesan overlay loaded.
+; - The single malformed trailing entry ("...The OCTAVE DAY of the
+; NATIVITY..." with a bare 4-digit token where a weekday+day-number
+; pair belongs) is excluded outright: the extractor's own stop
+; marker (a line whose first token is a bare 4-digit token) halts
+; the scan before it. CONFIRMED IDENTICAL, word for word, across
+; ALL FOUR known editions of this publisher's Ordo (2023-2024,
+; 2024-2025, 2025-2026, and the 2024-2025 fixture's own prior
+; header) -- always the literal stray year token "2025", regardless
+; of which edition or which real trailing year it should read: a
+; stale copy-paste leftover the compiler's own master document
+; carries forward unedited release to release (even the 2024-2025
+; PDF's own tail page footer nearby still reads "Latin Mass Society
+; Ordo 2021-2022"), not something specific to this edition.
+; - Front matter, the Abbreviations/Rubrical-Primer/Breviary sections,
+; and the "When may I say...?"/Appendix tail are never scanned at
+; all (the extractor's own start/stop markers bound it to the Ordo
+; proper).
+;
+; 5. THE "Pr of" (PREFACE) COLUMN (Preface-witnesses task, 2026-08-23) --
+; [row.praef], the RAW text from the FIRST "Pr of ..." / "Common Pr"
+; match in the universal block's own reading order (the same
+; FIRST-in-reading-order discipline finding 1 above already
+; establishes for Gl/Cr). CAPTURED, NOT CLASSIFIED by this tool --
+; classification into {!Colitur_kernel.Preface.t} and the comparison
+; against {!Rite_ef.Rubrics_ef.preface} both happen test-side
+; (test/test_lms_ordo.ml's own [classify_praef]).
+;
+; CHARACTERISATION FINDING, non-negotiable per the same discipline as
+; finding 2: this Ordo prints OPTION LISTS ("Pr of X or Pr of Y or
+; Common Pr"), not always a single value -- and several of the named
+; options ("Martyrs", "All Saints and Patron Saints" [almost always
+; diocesan -- a Patronal feast's own preface -- but checked directly to
+; occur ONCE in THIS edition's own UNIVERSAL block too, All Saints' Day
+; itself, 1 November, paired with "or Common Pr"/"or Pr of the
+; Trinity" depending on whether 1 November falls on a Sunday that
+; edition, exactly matching {!Rite_ef.Rubrics_ef.preface}'s own RG
+; 494(b) Sunday-gated Trinity fallback either way -- see that
+; function's own citation], "the Dedication of a Church", "the Most Holy
+; Sacrament" [Corpus Christi's own universal entry], "St John the
+; Baptist", "the Angels") are genuinely NOT among the fourteen the
+; 1962 Missale Romanum's own RG 484-497 enumerate (docs/research/LT.txt,
+; lines 3936-4020, checked directly: no "de Martyribus"/"de Angelis"/
+; "de Dedicatione"/"de Ss.mo Sacramento"/"de S. Ioanne Baptista" clause
+; exists anywhere in 482-499). Every one of these extras, in every
+; instance found in this edition's UNIVERSAL block, is printed
+; ALONGSIDE a genuine RG 482 answer (most often "or Common Pr"; on 24
+; March-shaped days, "or Pr of Lent", the live RG 486(b) de-Tempore
+; answer) -- never as the row's OWN AND ONLY option. This is why the
+; test-side comparison checks colitur's single computed answer is a
+; MEMBER of this row's own parsed option set (dropping any unmapped
+; extra name from that set first), not string equality against
+; whichever option happens to print first -- the same "capture, don't
+; force a match invented at extraction time" discipline
+; tools/extract_fiuv_ordo.ml's own [row.praef] doc comment already
+; states for a different publisher's format.
+;
+; OPPOSITE-PREDICTION PROBES against Rite_ef.Rubrics_ef.preface (RG
+; 482-499), the SAME discipline as finding 2, applied to this new
+; column -- RE-CHECKED BY THIS TOOL RUN before this header was
+; written; a failed probe is a hard [die]:
+%s
+; All %d predictions confirmed -- this column, too, is discriminating,
+; not a constant, across three different RG-cited outcomes (Nativity/
+; Apostles/Requiem).
+;
+; COVERAGE: %d of %d rows have no "Pr of"/"Common Pr" text found at
+; all -- checked directly against this window's own Good Friday
+; (Easter-2): its block carries NO Gl/Cr/Pr line whatsoever in this
+; source (the 1955-restored Holy Week's own liturgical action has no
+; Mass that day at all -- the SAME structural fact
+; {!test_creed_coverage}'s own [window_good_fridays] already asserts
+; for Gl/Cr), so [praef] shares that single expected gap rather than
+; having a coverage gap of its own.
+|}
+ dest edition_label isbn copyright_year pdf_meta url_note url pdf_base pdf_base pdf_sha (today ()) pdf_base
+ pdf_base dest
+ (List.length rows)
+ (match rows with r :: _ -> r.date | [] -> "?")
+ (match List.rev rows with r :: _ -> r.date | [] -> "?")
+ window_open_desc probe_lines (List.length probes) with_variant (List.length rows)
+ (Printf.sprintf "%.1f" (100.0 *. float_of_int with_variant /. float_of_int (List.length rows)))
+ praef_probe_lines (List.length praef_probes) missing_praef_count (List.length rows)
+ in
+ let body = Sexplib.Sexp.to_string_hum ~indent:2 (sexp_of_list sexp_of_row rows) in
+ let oc = open_out dest in
+ Fun.protect
+ ~finally:(fun () -> close_out_noerr oc)
+ (fun () ->
+ output_string oc header;
+ output_string oc body;
+ output_string oc "\n");
+ Printf.printf "extract_lms_ordo: wrote %d rows to %s\n" (List.length rows) dest