diff options
Diffstat (limited to 'tools')
| -rw-r--r-- | tools/extract_fiuv_ordo.ml | 221 |
1 files changed, 209 insertions, 12 deletions
diff --git a/tools/extract_fiuv_ordo.ml b/tools/extract_fiuv_ordo.ml index c9688a7..26f5b16 100644 --- a/tools/extract_fiuv_ordo.ml +++ b/tools/extract_fiuv_ordo.ml @@ -68,6 +68,40 @@ extract_lms_ordo.ml's own header describes for "V Mass of BVM" being a substring of "IV Mass of BVM". + 5. A WHOLE-SECOND-OFFICE TRAP, found and defended against + (celebrant-rubrics-phase1 Phase 2 correction, 2026-08-22, closing + data/ef/expected-divergences-fiuv.sexp's own F3): six real days + print BOTH a strict-1962 reading (a privileged Lenten/Passiontide + feria, saint reduced to a commemoration, "Off. feriale") AND, after + a literal separator token, a SECOND, complete alternative office + (the same saint kept outright under the 2020 decree Cum + Sanctissima, "Off. ordinarium", with its OWN "Ad Mat. ... Te Deum" + lessons). The separator is an en-dash, "VEL" (Latin "or"), another + en-dash, glued by pdftotext into ONE token with no internal spaces + -- "\xe2\x80\x93VEL\xe2\x80\x93" ("VEL" wrapped in en-dashes) -- + confirmed by direct byte inspection of the raw dump, NOT the same + shape as [extract_missa_fields]'s own "Vel"+"Missa" two-token + boundary above (a narrower "choice of Mass formulary within the + SAME office" marker, e.g. a Common vs a Proper Mass for one saint). + [extract_missa_fields]'s own Gloria/Credo/praef span happened to + stay correctly bounded to the PRIMARY office on all six real + instances even before this fix, because a Vespers line ("VESPER..." + -- already one of its own boundary tokens) always intervenes before + the separator in this corpus; [extract_te_deum] had NO such + protection at all -- it scans the WHOLE block's own text for "Ad + Mat." unbounded, so on these six days (whose PRIMARY office states + no "Ad Mat." of its own at all -- an ordinary privileged feria's + Matins needs no special note) it read the SECOND office's own "Ad + Mat. ... Te Deum." instead, manufacturing a same-day contradiction + that F3 first recorded as "colitur's own possible gap". [split_at_alt_office] + below cuts every block's own token array at this separator BEFORE + any field extraction runs, so every extractor now reads the PRIMARY + office only; the alternative is captured, not discarded, in its own + [alt_te_deum] field (the row type's own comment has the full + account of why only this one field, not a parallel alt_gloria/ + alt_credo/alt_class -- those three were never wrong, so widening + them now would be undirected scope creep, not a fix). + Usage: pdftotext -layout docs/research/ordo/fiuv-ordo-2025-2026.pdf /tmp/fiuv.txt dune exec tools/extract_fiuv_ordo.exe -- /tmp/fiuv.txt \ @@ -84,15 +118,46 @@ let die fmt = Printf.ksprintf (fun s -> prerr_endline ("extract_fiuv_ordo: " ^ s reasoning extract_lms_ordo.ml's own header already gives. *) type row = { date : string; (** ISO-8601 *) - class_ : string option; (** raw, e.g. "III cl.", "III cl. (Priv.)", "I cl." -- [None] only if genuinely unparseable *) - title : string; (** the day-start line's own text before the class marker *) - te_deum : bool option; (** [None] only if neither "Te Deum" nor "non dicitur Te Deum" is found *) - gloria : bool option; - credo : bool option; + class_ : string option; (** raw, e.g. "III cl.", "III cl. (Priv.)", "I cl." -- [None] only if genuinely unparseable, read off the PRIMARY office alone *) + title : string; (** the day-start line's own text before the class marker, PRIMARY office alone *) + te_deum : bool option; (** [None] only if neither "Te Deum" nor "non dicitur Te Deum"/"sine Te Deum" is found WITHIN THE PRIMARY OFFICE (see [split_at_alt_office]) *) + gloria : bool option; (** PRIMARY office *) + credo : bool option; (** PRIMARY office *) praef : string option; (** raw trailing text after "praef." within the primary Mass clause -- CAPTURED, NOT VALIDATED *) + alt_te_deum : bool option; + (** the SECOND, Cum-Sanctissima-alternative office's own Te Deum + status, when this day prints one (see [split_at_alt_office]'s + own citation) -- [None] both when the day has no such + alternative at all (394 of 400 real days) AND when it has one + but that alternative's own text doesn't state Te Deum either + (never observed live: on all six real instances the alternative + states it unnegated). Captured rather than discarded because a + FUTURE Cum-Sanctissima overlay (CLAUDE.md's own binding + decision 2 -- an addition, never core) would have exactly this + shape to compare against; not compared by anything in this + project yet. Deliberately NOT joined by an [alt_gloria]/ + [alt_credo]/[alt_class] -- see [split_at_alt_office]'s own + citation for why only this one field was ever wrong. *) } [@@deriving sexp] +(* The literal separator between a day's PRIMARY (strict 1962) office and + a printed SECOND, Cum-Sanctissima-2020 alternative -- see the module + header's own point 5 for the full account of the bug this closes + (data/ef/expected-divergences-fiuv.sexp's own F3) and why it is a + DIFFERENT token from [extract_missa_fields]'s own "Vel"+"Missa" + two-token boundary. Exactly 6 occurrences in the whole corpus, + confirmed by direct grep of the raw pdftotext dump -- all 6 on the + exact dates F3 named (2026-03-06/07/09/12/21/24). *) +let alt_office_marker = "\xe2\x80\x93VEL\xe2\x80\x93" + +let split_at_alt_office tokens = + let n = Array.length tokens in + let rec go i = if i >= n then None else if tokens.(i) = alt_office_marker then Some i else go (i + 1) in + match go 0 with + | None -> (tokens, None) + | Some i -> (Array.sub tokens 0 i, Some (Array.sub tokens (i + 1) (n - i - 1))) + (* --- tiny hand-rolled helpers, no Str/regex ------------------------------ *) let split_ws s = @@ -366,6 +431,29 @@ let extract_missa_fields tokens = let rec go i = if i >= n then n else if tokens.(i) = "Vel" && i + 1 < n && tokens.(i + 1) = "Missa" then i + (* A THIRD, narrower alternate-Mass marker, found while auditing + for the "-VEL-" trap above (celebrant-rubrics-phase1 Phase 2 + correction, 2026-08-22): "Vel (sec. decretum <<Cum sanct.>>) + Missa ..." offers an alternate MASS FORMULARY under the 2020 + decree Cum Sanctissima WITHIN the same single printed office + (a narrower thing than [split_at_alt_office]'s own full + second-office split above -- there is no second "Ad Mat." set + of lessons here, just a second Mass choice). Checked against + the raw dump directly: 20 occurrences, ALL 20 of the + capitalised "Vel" + open-paren shape, zero false positives + (the ordinary lower-case "vel" conjunction, e.g. "praef. + comm. vel de Martyribus", never starts with an open-paren + token, so this check never fires on it). Every one of the 20 + real days already restates Gloria/Credo explicitly BEFORE + reaching this marker, so this was NEVER a live divergence + (confirmed by a direct diff of every field this fixture + compares, before and after this hardening: zero rows + changed) -- added defensively, so a FUTURE year whose + primary Mass clause happens not to restate Gloria/Credo + before this marker cannot silently read the Cum-Sanctissima + alternative's own value instead, the identical failure shape + as the "-VEL-" bug above, just not yet observed live here. *) + else if tokens.(i) = "Vel" && i + 1 < n && has_prefix ~prefix:"(" tokens.(i + 1) then i else if has_prefix ~prefix:"VESPER" tokens.(i) then i else go (i + 1) in @@ -475,12 +563,20 @@ let () = dropped before field extraction, kept for nothing (the day number is already known from [b_day]). *) let content = if Array.length tokens > 0 then Array.sub tokens 1 (Array.length tokens - 1) else tokens in - let class_ = extract_class content in - let title = extract_title content in - let full_text = String.concat " " (Array.to_list content) in + (* Cut at the "–VEL–" alternative-office boundary, if this day has + one, BEFORE any field extraction -- see [split_at_alt_office]'s + own citation. Every field below now reads the PRIMARY office + only; [alt_te_deum] alone also reads the second, when present. *) + let primary, alt = split_at_alt_office content in + let class_ = extract_class primary in + let title = extract_title primary in + let full_text = String.concat " " (Array.to_list primary) in let te_deum = extract_te_deum full_text in - let gloria, credo, praef = extract_missa_fields content in - { date = Date.to_iso8601 date; class_; title; te_deum; gloria; credo; praef }) + let alt_te_deum = + match alt with None -> None | Some alt_tokens -> extract_te_deum (String.concat " " (Array.to_list alt_tokens)) + in + let gloria, credo, praef = extract_missa_fields primary in + { date = Date.to_iso8601 date; class_; title; te_deum; gloria; credo; praef; alt_te_deum }) blocks in let missing_class = List.filter (fun r -> r.class_ = None) rows in @@ -517,6 +613,22 @@ let () = | None -> die "characterisation probe FAILED: %s (%s) -- no row extracted for this date at all" d desc) probes; Printf.printf "extract_fiuv_ordo: all %d characterisation probes confirmed\n" (List.length probes); + (* Second characterisation probe, same discipline: the "–VEL–" + alternative-office split (module header point 5) must fire on + EXACTLY these six dates, no more, no fewer -- a hard [die], not a + printed claim, so a future re-extraction against a re-flowed dump + silently gaining or losing an instance fails loudly rather than + quietly changing F3's own population. *) + let alt_office_dates = List.filter_map (fun r -> if r.alt_te_deum <> None then Some r.date else None) rows in + let expected_alt_office_dates = + [ "2026-03-06"; "2026-03-07"; "2026-03-09"; "2026-03-12"; "2026-03-21"; "2026-03-24" ] + in + if List.sort String.compare alt_office_dates <> expected_alt_office_dates then + die "alt-office-split characterisation probe FAILED: expected exactly %s, got %s" + (String.concat ", " expected_alt_office_dates) + (String.concat ", " (List.sort String.compare alt_office_dates)); + Printf.printf "extract_fiuv_ordo: alt-office-split characterisation probe confirmed (%d dates)\n" + (List.length expected_alt_office_dates); let probe_lines = String.concat "\n" (List.map @@ -542,6 +654,9 @@ let () = let credo_false = List.length (List.filter (fun r -> r.credo = Some false) rows) in let te_deum_true = List.length (List.filter (fun r -> r.te_deum = Some true) rows) in let te_deum_false = List.length (List.filter (fun r -> r.te_deum = Some false) rows) in + let alt_te_deum_true = List.length (List.filter (fun r -> r.alt_te_deum = Some true) rows) in + let alt_te_deum_false = List.length (List.filter (fun r -> r.alt_te_deum = Some false) rows) in + let alt_te_deum_none = List.length (List.filter (fun r -> r.alt_te_deum = None) rows) in let header = Printf.sprintf {|; %s -- Witnesses task (2026-08-22-colitur- @@ -637,15 +752,97 @@ let () = ; below are what this fixture actually asserts, not a forced match to ; the brief's approximate figures. ; +; 5. THE ALTERNATIVE-OFFICE TRAP (celebrant-rubrics-phase1 Phase 2 +; correction, 2026-08-22, closing F3): six real days print BOTH a +; strict-1962 reading and a complete SECOND office (the same saint +; kept outright under the 2020 decree Cum Sanctissima), separated by +; a literal "-VEL-" token. [extract_te_deum] used to scan the WHOLE +; block unbounded, so on these six days -- whose PRIMARY office states +; no "Ad Mat." of its own -- it silently read the SECOND office's own +; Te Deum instead, manufacturing a same-day contradiction with colitur +; that F3 first recorded as "colitur's own possible gap". Every field +; now reads the PRIMARY office only (module header point 5 has the +; full account); the alternative is captured, not discarded, in its +; own [alt_te_deum] field. Confirmed to fire on EXACTLY six dates by a +; hard characterisation probe (this run's own output, above), not +; merely asserted here. +; ; MEASURED DISTRIBUTION (this extraction, not the brief's figures): Gloria ; true=%d false=%d unresolved=%d; Credo true=%d false=%d unresolved=%d; -; Te Deum true=%d false=%d unresolved=%d. +; Te Deum true=%d false=%d unresolved=%d; alt Te Deum (the six "-VEL-" +; Cum Sanctissima alternatives, point 5 above) true=%d false=%d none=%d +; (a day contributes to "none" here either because it has no "-VEL-" +; alternative at all -- 394 of 400 -- or, for a day that does, because +; that alternative's own text doesn't state Te Deum either; not +; distinguished by this one count). +; +; FIX HISTORY (fixed text, HISTORICAL counts from the run that found each +; bug, not re-verified live by every later run -- this section exists so +; regenerating this fixture from a fresh pdftotext dump, which rewrites +; this whole header from the template, does not silently erase either +; finding; earlier this header carried the first entry below as a +; hand-appended note that a regeneration between then and now had +; already dropped once): +; +; 1. RE-EXTRACTED (celebrant-rubrics-phase1 Phase 2, 2026-08-22): the FIRST +; extraction's own Te Deum parser recognised only ONE of the source's two +; negative phrasings ("non dicitur Te Deum") -- "sine Te Deum" (how the +; source actually negates a SUNDAY's own Te Deum, among others) fell +; through to a bare "Te Deum" substring match and was wrongly read +; [true]. Found by hand, comparing the raw pdftotext dump against this +; fixture's own values directly, not by any coverage check (both counted +; the same 127 unresolved rows before and after). Fixed in +; tools/extract_fiuv_ordo.ml's own [extract_te_deum] (see its own +; citation); this fixture re-extracted from the SAME pdftotext dump +; (SHA-256 of the source PDF unchanged, above). 24 of 400 rows changed, +; every one Te Deum true->false, every one independently confirmed +; against the raw text: true 231->207, false 42->66, unresolved 127 +; unchanged. See lib/rites/rite_ef/rubrics_ef.ml's own [te_deum] header +; (237(a)'s comment) for the full account of what this corrected, and +; the .superpowers task report for the complete before/after date list. +; +; 2. RE-EXTRACTED (celebrant-rubrics-phase1 Phase 2 correction, 2026-08-22): +; module header point 5 and CHARACTERISATION point 5 above have the +; full account of the "-VEL-" alternative-office bug this closes +; (data/ef/expected-divergences-fiuv.sexp's own F3). This fixture +; re-extracted from the SAME pdftotext dump (SHA-256 of the source PDF +; unchanged, above). 6 of 400 rows changed, every one Te Deum +; true->unresolved (2026-03-06/07/09/12/21/24), every one independently +; confirmed against the raw text: Te Deum true 207->201 (this run's own +; unresolved count went 127->133, consistent with those same 6 rows). +; Gloria/Credo/class_/title/praef are UNCHANGED on all 400 rows -- +; those fields were already correctly bounded to the primary office +; before this fix (a Vespers-line token always intervened before the +; "-VEL-" separator on all six real instances), confirmed by a direct +; diff of the fixture before and after, not merely by argument. +; +; 3. RE-EXTRACTED (celebrant-rubrics-phase1 same correction, defensive +; hardening, 2026-08-22): auditing for the SAME trap elsewhere (the +; task brief's own item 5) found a THIRD, narrower alternate-Mass +; marker, "Vel (sec. decretum <<Cum sanct.>>) Missa ...", offering a +; Cum-Sanctissima Mass FORMULARY choice within a single office +; (unlike point 5 above, no second "Ad Mat." set of lessons -- a +; narrower thing than a full alternate office). 20 occurrences in the +; whole corpus, confirmed by direct grep; on all 20, Gloria/Credo were +; ALREADY stated before reaching this marker, so this was NEVER a +; live divergence for those two fields (confirmed by this same diff: +; 0 Gloria/Credo/class_/title/alt_te_deum rows changed). [praef] +; alone changed on 20 of 400 rows (its own "rest of the primary +; clause" capture no longer runs past this marker into the +; alternative's own text) -- harmless in itself ([praef] is +; documented CAPTURED, NOT VALIDATED, compared by nothing), but added +; as a defensive boundary in [extract_missa_fields] so a FUTURE year +; whose primary clause does not restate Gloria/Credo before this +; marker cannot silently read the alternative's own value instead -- +; the identical failure shape as point 2 above, closed before it was +; ever observed live rather than after. |} dest pdf_base pdf_base pdf_sha (today ()) pdf_base pdf_base dest (List.length rows) (match rows with r :: _ -> r.date | [] -> "?") (match List.rev rows with r :: _ -> r.date | [] -> "?") probe_lines (List.length probes) gloria_true gloria_false (List.length missing_gloria) credo_true credo_false - (List.length missing_credo) te_deum_true te_deum_false (List.length missing_te_deum) + (List.length missing_credo) te_deum_true te_deum_false (List.length missing_te_deum) alt_te_deum_true + alt_te_deum_false alt_te_deum_none in let body = Sexplib.Sexp.to_string_hum ~indent:2 (sexp_of_list sexp_of_row rows) in let oc = open_out dest in |
